{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NAPJYZLSGEDJI66IDF3JBM5TGG","short_pith_number":"pith:NAPJYZLS","schema_version":"1.0","canonical_sha256":"681e9c65723106947bc8197690b3b331aca6dad48479e258ea0af268a665cd53","source":{"kind":"arxiv","id":"2501.13124","version":1},"attestation_state":"computed","paper":{"title":"Debate Helps Weak-to-Strong Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Hao Lang, Yongbin Li","submitted_at":"2025-01-21T05:36:13Z","abstract_excerpt":"Common methods for aligning already-capable models with desired behavior rely on the ability of humans to provide supervision. However, future superhuman models will surpass the capability of humans. Therefore, humans will only be able to weakly supervise superhuman models. This expected deficiency of human evaluation would weaken the safety of future AI systems. Scalable oversight and weak-to-strong generalization are two complementary approaches to tackle this issue. In this paper, we attempt to combine the strengths of these two approaches to further improve alignment. Specifically, we inve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.13124","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-21T05:36:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9e7eb7d16f3c591be335b2b62cf3e6cfbfe8b1cbdcdd0db500e1e6890d380b32","abstract_canon_sha256":"1297f30ef05010a0a8b00573fe19ee009faf8ee39790e6ad4524d7081ffce2b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:23.759135Z","signature_b64":"ULDhOr6L9Rp/pDXQpa/5IR+kEJkHeklvSUg3d8N+rj6mEsRQ4j/bb2iDw4ndydAHtp7C554alAd/MCIBLIWbCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"681e9c65723106947bc8197690b3b331aca6dad48479e258ea0af268a665cd53","last_reissued_at":"2026-07-05T10:04:23.758652Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:23.758652Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Debate Helps Weak-to-Strong Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Hao Lang, Yongbin Li","submitted_at":"2025-01-21T05:36:13Z","abstract_excerpt":"Common methods for aligning already-capable models with desired behavior rely on the ability of humans to provide supervision. However, future superhuman models will surpass the capability of humans. Therefore, humans will only be able to weakly supervise superhuman models. This expected deficiency of human evaluation would weaken the safety of future AI systems. Scalable oversight and weak-to-strong generalization are two complementary approaches to tackle this issue. In this paper, we attempt to combine the strengths of these two approaches to further improve alignment. Specifically, we inve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.13124","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.13124/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.13124","created_at":"2026-07-05T10:04:23.758706+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.13124v1","created_at":"2026-07-05T10:04:23.758706+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.13124","created_at":"2026-07-05T10:04:23.758706+00:00"},{"alias_kind":"pith_short_12","alias_value":"NAPJYZLSGEDJ","created_at":"2026-07-05T10:04:23.758706+00:00"},{"alias_kind":"pith_short_16","alias_value":"NAPJYZLSGEDJI66I","created_at":"2026-07-05T10:04:23.758706+00:00"},{"alias_kind":"pith_short_8","alias_value":"NAPJYZLS","created_at":"2026-07-05T10:04:23.758706+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05710","citing_title":"On the Blessing of Pre-training in Weak-to-Strong Generalization","ref_index":117,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG","json":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG.json","graph_json":"https://pith.science/api/pith-number/NAPJYZLSGEDJI66IDF3JBM5TGG/graph.json","events_json":"https://pith.science/api/pith-number/NAPJYZLSGEDJI66IDF3JBM5TGG/events.json","paper":"https://pith.science/paper/NAPJYZLS"},"agent_actions":{"view_html":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG","download_json":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG.json","view_paper":"https://pith.science/paper/NAPJYZLS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.13124&json=true","fetch_graph":"https://pith.science/api/pith-number/NAPJYZLSGEDJI66IDF3JBM5TGG/graph.json","fetch_events":"https://pith.science/api/pith-number/NAPJYZLSGEDJI66IDF3JBM5TGG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG/action/storage_attestation","attest_author":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG/action/author_attestation","sign_citation":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG/action/citation_signature","submit_replication":"https://pith.science/pith/NAPJYZLSGEDJI66IDF3JBM5TGG/action/replication_record"}},"created_at":"2026-07-05T10:04:23.758706+00:00","updated_at":"2026-07-05T10:04:23.758706+00:00"}