{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6MLAATE2ADE255LMLCIFBX6SRP","short_pith_number":"pith:6MLAATE2","schema_version":"1.0","canonical_sha256":"f316004c9a00c9aef56c589050dfd28bd792121bbfa1e9ee2c42578de6132bb7","source":{"kind":"arxiv","id":"2404.19733","version":3},"attestation_state":"computed","paper":{"title":"Iterative Reasoning Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"He He, Jason Weston, Kyunghyun Cho, Richard Yuanzhe Pang, Sainbayar Sukhbaatar, Weizhe Yuan","submitted_at":"2024-04-30T17:28:05Z","abstract_excerpt":"Iterative preference optimization methods have recently been shown to perform well for general instruction tuning tasks, but typically make little improvement on reasoning tasks (Yuan et al., 2024, Chen et al., 2024). In this work we develop an iterative approach that optimizes the preference between competing generated Chain-of-Thought (CoT) candidates by optimizing for winning vs. losing reasoning steps that lead to the correct answer. We train using a modified DPO loss (Rafailov et al., 2023) with an additional negative log-likelihood term, which we find to be crucial. We show reasoning imp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.19733","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-30T17:28:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"884037fe077cae6aeaf369e705f8f8609bc8d138b6d108cccf22b084e4f85fe4","abstract_canon_sha256":"301c9246dc063e24bf57bd6248ac62a669f315d2934a54928b8d12d58ce38c08"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:47.468702Z","signature_b64":"M2C+gEoi/fPCojxUK5hhUtVvgvRh5ey+C10Oud6fE/rFuvC9dduCuhtG9yIVLSnelUgnYMvFbvqa0VEjPkLDBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f316004c9a00c9aef56c589050dfd28bd792121bbfa1e9ee2c42578de6132bb7","last_reissued_at":"2026-07-05T08:36:47.468236Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:47.468236Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Iterative Reasoning Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"He He, Jason Weston, Kyunghyun Cho, Richard Yuanzhe Pang, Sainbayar Sukhbaatar, Weizhe Yuan","submitted_at":"2024-04-30T17:28:05Z","abstract_excerpt":"Iterative preference optimization methods have recently been shown to perform well for general instruction tuning tasks, but typically make little improvement on reasoning tasks (Yuan et al., 2024, Chen et al., 2024). In this work we develop an iterative approach that optimizes the preference between competing generated Chain-of-Thought (CoT) candidates by optimizing for winning vs. losing reasoning steps that lead to the correct answer. We train using a modified DPO loss (Rafailov et al., 2023) with an additional negative log-likelihood term, which we find to be crucial. We show reasoning imp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.19733","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.19733/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.19733","created_at":"2026-07-05T08:36:47.468290+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.19733v3","created_at":"2026-07-05T08:36:47.468290+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.19733","created_at":"2026-07-05T08:36:47.468290+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MLAATE2ADE2","created_at":"2026-07-05T08:36:47.468290+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MLAATE2ADE255LM","created_at":"2026-07-05T08:36:47.468290+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MLAATE2","created_at":"2026-07-05T08:36:47.468290+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21090","citing_title":"Self-Improvement Can Self-Regress: The Rise-and-Collapse Failure Mode of LLM Self-Training","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":186,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28998","citing_title":"Reward-Free Code Alignment from Pretrained or Fine-Tuned LLM: Unpacking the Trade-offs for Code Generation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29150","citing_title":"Flow Reasoning Models: Scaling Reasoning Through Iterative Self-Refinement","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2502.00955","citing_title":"Efficient Multi-Agent System Training with Data Influence-Oriented Tree Search","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11574","citing_title":"Learning to Configure Agentic AI Systems","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11574","citing_title":"Learning to Configure Agentic AI Systems","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13228","citing_title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2502.03387","citing_title":"LIMO: Less is More for Reasoning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18512","citing_title":"S2H-DPO: Hardness-Aware Preference Optimization for Vision-Language Models","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP","json":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP.json","graph_json":"https://pith.science/api/pith-number/6MLAATE2ADE255LMLCIFBX6SRP/graph.json","events_json":"https://pith.science/api/pith-number/6MLAATE2ADE255LMLCIFBX6SRP/events.json","paper":"https://pith.science/paper/6MLAATE2"},"agent_actions":{"view_html":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP","download_json":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP.json","view_paper":"https://pith.science/paper/6MLAATE2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.19733&json=true","fetch_graph":"https://pith.science/api/pith-number/6MLAATE2ADE255LMLCIFBX6SRP/graph.json","fetch_events":"https://pith.science/api/pith-number/6MLAATE2ADE255LMLCIFBX6SRP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP/action/storage_attestation","attest_author":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP/action/author_attestation","sign_citation":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP/action/citation_signature","submit_replication":"https://pith.science/pith/6MLAATE2ADE255LMLCIFBX6SRP/action/replication_record"}},"created_at":"2026-07-05T08:36:47.468290+00:00","updated_at":"2026-07-05T08:36:47.468290+00:00"}