{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SJ4L3HVG7FJB72H6H7KPD3FM2B","short_pith_number":"pith:SJ4L3HVG","schema_version":"1.0","canonical_sha256":"9278bd9ea6f9521fe8fe3fd4f1ecacd0498df5bd52022cc02c2acfd46055819d","source":{"kind":"arxiv","id":"2407.00215","version":1},"attestation_state":"computed","paper":{"title":"LLM Critics Help Catch LLM Bugs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Evgenia Nitishinskaya, Jan Leike, Juan Felipe Ceron Uribe, Maja Trebacz, Nat McAleese, Rai Michael Pokorny","submitted_at":"2024-06-28T19:53:17Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is fundamentally limited by the capacity of humans to correctly evaluate model output. To improve human evaluation ability and overcome that limitation this work trains \"critic\" models that help humans to more accurately evaluate model-written code. These critics are themselves LLMs trained with RLHF to write natural language feedback highlighting problems in code from real-world assistant tasks. On code containing naturally occurring LLM errors model-written critiques are preferred over human critiques in 63% of cases, and human evaluation fin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00215","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-06-28T19:53:17Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e48f952d676c13f40b0776947d8a735d7c08d415f0c29f5dc664987c86f35e2c","abstract_canon_sha256":"a60463404cec2694df7b456c1d6ffcb76a64efdef74d786786bba83e6fbe8fa3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:15.895233Z","signature_b64":"ZwJZrRvbZtpFSfzRChcygPB1lFHrssyjAMkOiC+MdReE6j7yzwb6VQRrBue2/enPT6d4KelKNwlEBCqNOh1GAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9278bd9ea6f9521fe8fe3fd4f1ecacd0498df5bd52022cc02c2acfd46055819d","last_reissued_at":"2026-07-05T08:38:15.894812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:15.894812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM Critics Help Catch LLM Bugs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Evgenia Nitishinskaya, Jan Leike, Juan Felipe Ceron Uribe, Maja Trebacz, Nat McAleese, Rai Michael Pokorny","submitted_at":"2024-06-28T19:53:17Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is fundamentally limited by the capacity of humans to correctly evaluate model output. To improve human evaluation ability and overcome that limitation this work trains \"critic\" models that help humans to more accurately evaluate model-written code. These critics are themselves LLMs trained with RLHF to write natural language feedback highlighting problems in code from real-world assistant tasks. On code containing naturally occurring LLM errors model-written critiques are preferred over human critiques in 63% of cases, and human evaluation fin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00215","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00215","created_at":"2026-07-05T08:38:15.894867+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00215v1","created_at":"2026-07-05T08:38:15.894867+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00215","created_at":"2026-07-05T08:38:15.894867+00:00"},{"alias_kind":"pith_short_12","alias_value":"SJ4L3HVG7FJB","created_at":"2026-07-05T08:38:15.894867+00:00"},{"alias_kind":"pith_short_16","alias_value":"SJ4L3HVG7FJB72H6","created_at":"2026-07-05T08:38:15.894867+00:00"},{"alias_kind":"pith_short_8","alias_value":"SJ4L3HVG","created_at":"2026-07-05T08:38:15.894867+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00424","citing_title":"Weak Critics Make Strong Learners: On-Policy Critique Distillation for Scalable Oversight","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23108","citing_title":"Philosophical Dispositions as Behavioral Constraints for AI-Assisted Code Review: An Empirical Study","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15698","citing_title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2507.11473","citing_title":"Chain of Thought Monitorability: A New and Fragile Opportunity for AI Safety","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14311","citing_title":"Beyond Binary: Reframing GUI Critique as Continuous Semantic Alignment","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2601.06794","citing_title":"No More Stale Feedback: Co-Evolving Critics for Open-World Agent Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14311","citing_title":"Beyond Binary: Reframing GUI Critique as Continuous Semantic Alignment","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08321","citing_title":"LLM Wardens: Mitigating Adversarial Persuasion with Third-Party Conversational Oversight","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01643","citing_title":"AI Alignment via Incentives and Correction","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01643","citing_title":"AI Alignment via Incentives and Correction","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21718","citing_title":"Building a Precise Video Language with Human-AI Oversight","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07341","citing_title":"ReCodeAgent: A Multi-Agent Workflow for Language-agnostic Translation and Validation of Large-scale Repositories","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24955","citing_title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B","json":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B.json","graph_json":"https://pith.science/api/pith-number/SJ4L3HVG7FJB72H6H7KPD3FM2B/graph.json","events_json":"https://pith.science/api/pith-number/SJ4L3HVG7FJB72H6H7KPD3FM2B/events.json","paper":"https://pith.science/paper/SJ4L3HVG"},"agent_actions":{"view_html":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B","download_json":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B.json","view_paper":"https://pith.science/paper/SJ4L3HVG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00215&json=true","fetch_graph":"https://pith.science/api/pith-number/SJ4L3HVG7FJB72H6H7KPD3FM2B/graph.json","fetch_events":"https://pith.science/api/pith-number/SJ4L3HVG7FJB72H6H7KPD3FM2B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B/action/storage_attestation","attest_author":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B/action/author_attestation","sign_citation":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B/action/citation_signature","submit_replication":"https://pith.science/pith/SJ4L3HVG7FJB72H6H7KPD3FM2B/action/replication_record"}},"created_at":"2026-07-05T08:38:15.894867+00:00","updated_at":"2026-07-05T08:38:15.894867+00:00"}