{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KCVFRKJGYQDACDCQD2EFTGIAS4","short_pith_number":"pith:KCVFRKJG","schema_version":"1.0","canonical_sha256":"50aa58a926c406010c501e88599900972033d536fbdc9d043f9727da27a30beb","source":{"kind":"arxiv","id":"2406.04274","version":1},"attestation_state":"computed","paper":{"title":"Self-Play with Adversarial Critic: Provable and Scalable Offline Alignment for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Sanjeev Kulkarni, Tengyang Xie, Xiang Ji","submitted_at":"2024-06-06T17:23:49Z","abstract_excerpt":"This work studies the challenge of aligning large language models (LLMs) with offline preference data. We focus on alignment by Reinforcement Learning from Human Feedback (RLHF) in particular. While popular preference optimization methods exhibit good empirical performance in practice, they are not theoretically guaranteed to converge to the optimal policy and can provably fail when the data coverage is sparse by classical offline reinforcement learning (RL) results. On the other hand, a recent line of work has focused on theoretically motivated preference optimization methods with provable gu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.04274","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-06T17:23:49Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"11962222072524c5286d2866c27ebab45e0d1b16791db5534d6188cdf1e7a683","abstract_canon_sha256":"960330e3f31ca999bd316fa12ee03ad053063751813aa27371a4433acf470d4e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:28.809420Z","signature_b64":"qI1TrXz2F+vOAgF/AK2qt3CRvgqVe657uXqi3bDyW3nlMQwGV+HcrtSS75OMhFUOX/EKvd+JM8RkfzVm5cSPCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50aa58a926c406010c501e88599900972033d536fbdc9d043f9727da27a30beb","last_reissued_at":"2026-07-05T08:28:28.809000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:28.809000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Play with Adversarial Critic: Provable and Scalable Offline Alignment for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Sanjeev Kulkarni, Tengyang Xie, Xiang Ji","submitted_at":"2024-06-06T17:23:49Z","abstract_excerpt":"This work studies the challenge of aligning large language models (LLMs) with offline preference data. We focus on alignment by Reinforcement Learning from Human Feedback (RLHF) in particular. While popular preference optimization methods exhibit good empirical performance in practice, they are not theoretically guaranteed to converge to the optimal policy and can provably fail when the data coverage is sparse by classical offline reinforcement learning (RL) results. On the other hand, a recent line of work has focused on theoretically motivated preference optimization methods with provable gu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.04274","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.04274/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.04274","created_at":"2026-07-05T08:28:28.809054+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.04274v1","created_at":"2026-07-05T08:28:28.809054+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04274","created_at":"2026-07-05T08:28:28.809054+00:00"},{"alias_kind":"pith_short_12","alias_value":"KCVFRKJGYQDA","created_at":"2026-07-05T08:28:28.809054+00:00"},{"alias_kind":"pith_short_16","alias_value":"KCVFRKJGYQDACDCQ","created_at":"2026-07-05T08:28:28.809054+00:00"},{"alias_kind":"pith_short_8","alias_value":"KCVFRKJG","created_at":"2026-07-05T08:28:28.809054+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18531","citing_title":"When Does Trajectory-Level Supervision Permit Efficient Offline Reinforcement Learning?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01561","citing_title":"S-SPPO: Semantic-Calibrated Self-Play Preference Optimization","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20933","citing_title":"IRIS: Interpolative R\\'enyi Iterative Self-play for Large Language Model Fine-Tuning","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4","json":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4.json","graph_json":"https://pith.science/api/pith-number/KCVFRKJGYQDACDCQD2EFTGIAS4/graph.json","events_json":"https://pith.science/api/pith-number/KCVFRKJGYQDACDCQD2EFTGIAS4/events.json","paper":"https://pith.science/paper/KCVFRKJG"},"agent_actions":{"view_html":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4","download_json":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4.json","view_paper":"https://pith.science/paper/KCVFRKJG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.04274&json=true","fetch_graph":"https://pith.science/api/pith-number/KCVFRKJGYQDACDCQD2EFTGIAS4/graph.json","fetch_events":"https://pith.science/api/pith-number/KCVFRKJGYQDACDCQD2EFTGIAS4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4/action/storage_attestation","attest_author":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4/action/author_attestation","sign_citation":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4/action/citation_signature","submit_replication":"https://pith.science/pith/KCVFRKJGYQDACDCQD2EFTGIAS4/action/replication_record"}},"created_at":"2026-07-05T08:28:28.809054+00:00","updated_at":"2026-07-05T08:28:28.809054+00:00"}