{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2GTNA4RG76764POQGMLSMD2A36","short_pith_number":"pith:2GTNA4RG","schema_version":"1.0","canonical_sha256":"d1a6d07226ffbfee3dd03317260f40df94089a74f46656b8ab2c70f169b3406f","source":{"kind":"arxiv","id":"2404.12358","version":2},"attestation_state":"computed","paper":{"title":"From $r$ to $Q^*$: Your Language Model is Secretly a Q-Function","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chelsea Finn, Joey Hejna, Rafael Rafailov, Ryan Park","submitted_at":"2024-04-18T17:37:02Z","abstract_excerpt":"Reinforcement Learning From Human Feedback (RLHF) has been critical to the success of the latest generation of generative AI models. In response to the complex nature of the classical RLHF pipeline, direct alignment algorithms such as Direct Preference Optimization (DPO) have emerged as an alternative approach. Although DPO solves the same objective as the standard RLHF setup, there is a mismatch between the two approaches. Standard RLHF deploys reinforcement learning in a specific token-level MDP, while DPO is derived as a bandit problem in which the whole response of the model is treated as "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.12358","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-18T17:37:02Z","cross_cats_sorted":[],"title_canon_sha256":"5195bb7aa7c241bf2b2775c46586ffad6c666192fe4c1ec422cb11d8e4cf8cdb","abstract_canon_sha256":"88b1d8a2dac943e131ddef2e689973100498bb9627bb0139f5ad41b4e76a0a62"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:49.392917Z","signature_b64":"6ztiMd5XltOoI26QvYm+5lpucLZIdJkFZqXibLZI/pkbQGgMx0nDZXq5YKYWEMCqb5VJUfYe4PImt2AdqKPICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1a6d07226ffbfee3dd03317260f40df94089a74f46656b8ab2c70f169b3406f","last_reissued_at":"2026-07-05T08:54:49.392431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:49.392431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From $r$ to $Q^*$: Your Language Model is Secretly a Q-Function","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chelsea Finn, Joey Hejna, Rafael Rafailov, Ryan Park","submitted_at":"2024-04-18T17:37:02Z","abstract_excerpt":"Reinforcement Learning From Human Feedback (RLHF) has been critical to the success of the latest generation of generative AI models. In response to the complex nature of the classical RLHF pipeline, direct alignment algorithms such as Direct Preference Optimization (DPO) have emerged as an alternative approach. Although DPO solves the same objective as the standard RLHF setup, there is a mismatch between the two approaches. Standard RLHF deploys reinforcement learning in a specific token-level MDP, while DPO is derived as a bandit problem in which the whole response of the model is treated as "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12358","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.12358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.12358","created_at":"2026-07-05T08:54:49.392490+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.12358v2","created_at":"2026-07-05T08:54:49.392490+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12358","created_at":"2026-07-05T08:54:49.392490+00:00"},{"alias_kind":"pith_short_12","alias_value":"2GTNA4RG7676","created_at":"2026-07-05T08:54:49.392490+00:00"},{"alias_kind":"pith_short_16","alias_value":"2GTNA4RG76764POQ","created_at":"2026-07-05T08:54:49.392490+00:00"},{"alias_kind":"pith_short_8","alias_value":"2GTNA4RG","created_at":"2026-07-05T08:54:49.392490+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25582","citing_title":"Extreme Region Policy Distillation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2408.15339","citing_title":"UNA: A Unified Supervised Framework for Efficient LLM Alignment Across Feedback Types","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2411.00361","citing_title":"Direct Preference Optimization for Primitive-Enabled Hierarchical RL: A Bilevel Approach","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20218","citing_title":"Fine-grained List-wise Alignment for Generative Medication Recommendation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07832","citing_title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20408","citing_title":"Spectral Souping: A Unified Framework for Online Preference Alignment","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20265","citing_title":"Failure Modes of Maximum Entropy RLHF","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13228","citing_title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2410.18451","citing_title":"Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08378","citing_title":"Reinforcement Learning for Scalable and Trustworthy Intelligent Systems","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01456","citing_title":"Process Reinforcement through Implicit Rewards","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04874","citing_title":"Uncertainty-Aware Exploratory Direct Preference Optimization for Multimodal Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09459","citing_title":"From Reasoning to Agentic: Credit Assignment in Reinforcement Learning for Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15577","citing_title":"Reward Weighted Classifier-Free Guidance as Policy Improvement in Autoregressive Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02141","citing_title":"On the Optimal Sample Complexity of Offline Multi-Armed Bandits with KL Regularization","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36","json":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36.json","graph_json":"https://pith.science/api/pith-number/2GTNA4RG76764POQGMLSMD2A36/graph.json","events_json":"https://pith.science/api/pith-number/2GTNA4RG76764POQGMLSMD2A36/events.json","paper":"https://pith.science/paper/2GTNA4RG"},"agent_actions":{"view_html":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36","download_json":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36.json","view_paper":"https://pith.science/paper/2GTNA4RG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.12358&json=true","fetch_graph":"https://pith.science/api/pith-number/2GTNA4RG76764POQGMLSMD2A36/graph.json","fetch_events":"https://pith.science/api/pith-number/2GTNA4RG76764POQGMLSMD2A36/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/action/storage_attestation","attest_author":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/action/author_attestation","sign_citation":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/action/citation_signature","submit_replication":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/action/replication_record"}},"created_at":"2026-07-05T08:54:49.392490+00:00","updated_at":"2026-07-05T08:54:49.392490+00:00"}