{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VNS4EQHDGFKEJ3UWRELHYYDEWE","short_pith_number":"pith:VNS4EQHD","schema_version":"1.0","canonical_sha256":"ab65c240e3315444ee9689167c6064b122a5ae9a39f5e62eda952cb872dd4a8c","source":{"kind":"arxiv","id":"2311.08702","version":1},"attestation_state":"computed","paper":{"title":"Debate Helps Supervise Unreliable Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"David Rein, Jackson Petty, Julian Michael, Julien Dirani, Salsabila Mahdi, Samuel R. Bowman, Vishakh Padmakumar","submitted_at":"2023-11-15T05:05:40Z","abstract_excerpt":"As AI systems are used to answer more difficult questions and potentially help create new knowledge, judging the truthfulness of their outputs becomes more difficult and more important. How can we supervise unreliable experts, which have access to the truth but may not accurately report it, to give answers that are systematically true and don't just superficially seem true, when the supervisor can't tell the difference between the two on their own? In this work, we show that debate between two unreliable experts can help a non-expert judge more reliably identify the truth. We collect a dataset"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.08702","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-11-15T05:05:40Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"3ded429e4b98ca792a6a7bb11fd1a0d1d5416ae3cdc2ed0366cdbf7938bc4358","abstract_canon_sha256":"a0e09171aafc16f3953e8b9fba1b0ea584fec727f2479df1208875a9c06426ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:13:02.027832Z","signature_b64":"/BQJjjPhsVgh77fOFA4zb/PSEl+oqr/NkluYLQs6Ca74So+cAPV9elKb9iDmjoDQIKdMkHK5N8qWDW2ck263DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab65c240e3315444ee9689167c6064b122a5ae9a39f5e62eda952cb872dd4a8c","last_reissued_at":"2026-07-05T07:13:02.027348Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:13:02.027348Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Debate Helps Supervise Unreliable Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"David Rein, Jackson Petty, Julian Michael, Julien Dirani, Salsabila Mahdi, Samuel R. Bowman, Vishakh Padmakumar","submitted_at":"2023-11-15T05:05:40Z","abstract_excerpt":"As AI systems are used to answer more difficult questions and potentially help create new knowledge, judging the truthfulness of their outputs becomes more difficult and more important. How can we supervise unreliable experts, which have access to the truth but may not accurately report it, to give answers that are systematically true and don't just superficially seem true, when the supervisor can't tell the difference between the two on their own? In this work, we show that debate between two unreliable experts can help a non-expert judge more reliably identify the truth. We collect a dataset"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.08702","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.08702/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.08702","created_at":"2026-07-05T07:13:02.027407+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.08702v1","created_at":"2026-07-05T07:13:02.027407+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.08702","created_at":"2026-07-05T07:13:02.027407+00:00"},{"alias_kind":"pith_short_12","alias_value":"VNS4EQHDGFKE","created_at":"2026-07-05T07:13:02.027407+00:00"},{"alias_kind":"pith_short_16","alias_value":"VNS4EQHDGFKEJ3UW","created_at":"2026-07-05T07:13:02.027407+00:00"},{"alias_kind":"pith_short_8","alias_value":"VNS4EQHD","created_at":"2026-07-05T07:13:02.027407+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25899","citing_title":"Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05330","citing_title":"A Model of Multi-turn Human Persuadability Using Probabilistic Belief Tracing","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25133","citing_title":"Trust but Verify: Prover-Verifier Deliberation for Selective LLM Prediction","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17247","citing_title":"Towards Robust Argumentative Essay Understanding via TIDE: An Interactive Framework with Trial and Debate","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2402.05070","citing_title":"A Roadmap to Pluralistic Alignment","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04944","citing_title":"Inclusion-of-Thoughts: Mitigating Preference Instability via Purifying the Decision Space","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20131","citing_title":"Whose Story Gets Told? Positionality and Bias in LLM Summaries of Life Narratives","ref_index":155,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE","json":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE.json","graph_json":"https://pith.science/api/pith-number/VNS4EQHDGFKEJ3UWRELHYYDEWE/graph.json","events_json":"https://pith.science/api/pith-number/VNS4EQHDGFKEJ3UWRELHYYDEWE/events.json","paper":"https://pith.science/paper/VNS4EQHD"},"agent_actions":{"view_html":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE","download_json":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE.json","view_paper":"https://pith.science/paper/VNS4EQHD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.08702&json=true","fetch_graph":"https://pith.science/api/pith-number/VNS4EQHDGFKEJ3UWRELHYYDEWE/graph.json","fetch_events":"https://pith.science/api/pith-number/VNS4EQHDGFKEJ3UWRELHYYDEWE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE/action/storage_attestation","attest_author":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE/action/author_attestation","sign_citation":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE/action/citation_signature","submit_replication":"https://pith.science/pith/VNS4EQHDGFKEJ3UWRELHYYDEWE/action/replication_record"}},"created_at":"2026-07-05T07:13:02.027407+00:00","updated_at":"2026-07-05T07:13:02.027407+00:00"}