{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KNONEO3VQWQB32AIZQRXGHSFWA","short_pith_number":"pith:KNONEO3V","schema_version":"1.0","canonical_sha256":"535cd23b7585a01de808cc23731e45b01d0d0485a9930e635d05bca199b87b26","source":{"kind":"arxiv","id":"2501.03884","version":4},"attestation_state":"computed","paper":{"title":"AlphaPO: Reward Shape Matters for LLM Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aman Gupta, Ankan Saha, Eunki Kim, Jiwoo Hong, Natesh Pillai, Noah Lee, Parag Agrawal, Qingquan Song, Shao Tang, Sirou Zhu, Siyu Zhu, S. Sathiya Keerthi, Viral Gupta","submitted_at":"2025-01-07T15:46:42Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) and its variants have made huge strides toward the effective alignment of large language models (LLMs) to follow instructions and reflect human values. More recently, Direct Alignment Algorithms (DAAs) have emerged in which the reward modeling stage of RLHF is skipped by characterizing the reward directly as a function of the policy being learned. Some popular examples of DAAs include Direct Preference Optimization (DPO) and Simple Preference Optimization (SimPO). These methods often suffer from likelihood displacement, a phenomenon by which th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03884","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-07T15:46:42Z","cross_cats_sorted":[],"title_canon_sha256":"648eb80fca97ce99931b139b36db34dedf76b5aa7cef067fe9b706db912c660c","abstract_canon_sha256":"fe0430164b87c74c574dc48c3478903eecca7b3c91d4c95a93170fb204e1dff3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:24.157637Z","signature_b64":"HpY+8u2C5KeMC8jY4rSxV4uRy0IhKYpUySxQWhOviK1Z+4ZUYxFSsTissJM2tZ64hLl+xZfy3QZGkW9phbJ+AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"535cd23b7585a01de808cc23731e45b01d0d0485a9930e635d05bca199b87b26","last_reissued_at":"2026-07-05T11:12:24.157077Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:24.157077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AlphaPO: Reward Shape Matters for LLM Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aman Gupta, Ankan Saha, Eunki Kim, Jiwoo Hong, Natesh Pillai, Noah Lee, Parag Agrawal, Qingquan Song, Shao Tang, Sirou Zhu, Siyu Zhu, S. Sathiya Keerthi, Viral Gupta","submitted_at":"2025-01-07T15:46:42Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) and its variants have made huge strides toward the effective alignment of large language models (LLMs) to follow instructions and reflect human values. More recently, Direct Alignment Algorithms (DAAs) have emerged in which the reward modeling stage of RLHF is skipped by characterizing the reward directly as a function of the policy being learned. Some popular examples of DAAs include Direct Preference Optimization (DPO) and Simple Preference Optimization (SimPO). These methods often suffer from likelihood displacement, a phenomenon by which th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03884","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03884/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03884","created_at":"2026-07-05T11:12:24.157137+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03884v4","created_at":"2026-07-05T11:12:24.157137+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03884","created_at":"2026-07-05T11:12:24.157137+00:00"},{"alias_kind":"pith_short_12","alias_value":"KNONEO3VQWQB","created_at":"2026-07-05T11:12:24.157137+00:00"},{"alias_kind":"pith_short_16","alias_value":"KNONEO3VQWQB32AI","created_at":"2026-07-05T11:12:24.157137+00:00"},{"alias_kind":"pith_short_8","alias_value":"KNONEO3V","created_at":"2026-07-05T11:12:24.157137+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.20265","citing_title":"Failure Modes of Maximum Entropy RLHF","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA","json":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA.json","graph_json":"https://pith.science/api/pith-number/KNONEO3VQWQB32AIZQRXGHSFWA/graph.json","events_json":"https://pith.science/api/pith-number/KNONEO3VQWQB32AIZQRXGHSFWA/events.json","paper":"https://pith.science/paper/KNONEO3V"},"agent_actions":{"view_html":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA","download_json":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA.json","view_paper":"https://pith.science/paper/KNONEO3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03884&json=true","fetch_graph":"https://pith.science/api/pith-number/KNONEO3VQWQB32AIZQRXGHSFWA/graph.json","fetch_events":"https://pith.science/api/pith-number/KNONEO3VQWQB32AIZQRXGHSFWA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA/action/storage_attestation","attest_author":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA/action/author_attestation","sign_citation":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA/action/citation_signature","submit_replication":"https://pith.science/pith/KNONEO3VQWQB32AIZQRXGHSFWA/action/replication_record"}},"created_at":"2026-07-05T11:12:24.157137+00:00","updated_at":"2026-07-05T11:12:24.157137+00:00"}