{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5RPKB5NPJNIZHS2ZZSMSNLE5JZ","short_pith_number":"pith:5RPKB5NP","schema_version":"1.0","canonical_sha256":"ec5ea0f5af4b5193cb59cc9926ac9d4e57b5349484cb828648ed1bb9397600e2","source":{"kind":"arxiv","id":"2412.16145","version":2},"attestation_state":"computed","paper":{"title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"HanZe Dong, Huaijie Wang, Shenao Zhang, Shibo Hao, Yilin Bao, Yi Wu, Ziran Yang","submitted_at":"2024-12-20T18:49:45Z","abstract_excerpt":"Improving the multi-step reasoning ability of large language models (LLMs) with offline reinforcement learning (RL) is essential for quickly adapting them to complex tasks. While Direct Preference Optimization (DPO) has shown promise in aligning LLMs with human preferences, it is less suitable for multi-step reasoning tasks because (1) DPO relies on paired preference data, which is not readily available for multi-step reasoning tasks, and (2) it treats all tokens uniformly, making it ineffective for credit assignment in multi-step reasoning tasks, which often come with sparse reward. In this w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16145","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-20T18:49:45Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"252e1a4b0e2b81b48dc6d5efcd62589bce208a00ee86ae052bc26ecf0566379d","abstract_canon_sha256":"44ba39230a3ff895141c1dadbf231b8584e28e9ee0f585398f4be130bd8c5824"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:07.108244Z","signature_b64":"0jHEHoV77p3hqYd+HzAObsKKbXdhsbjCOUzHxg1pth8jco0ad1qag4mYmUgX02DPDyxDmTnO5B4NLfL/4I2SDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec5ea0f5af4b5193cb59cc9926ac9d4e57b5349484cb828648ed1bb9397600e2","last_reissued_at":"2026-07-05T09:54:07.107746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:07.107746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"HanZe Dong, Huaijie Wang, Shenao Zhang, Shibo Hao, Yilin Bao, Yi Wu, Ziran Yang","submitted_at":"2024-12-20T18:49:45Z","abstract_excerpt":"Improving the multi-step reasoning ability of large language models (LLMs) with offline reinforcement learning (RL) is essential for quickly adapting them to complex tasks. While Direct Preference Optimization (DPO) has shown promise in aligning LLMs with human preferences, it is less suitable for multi-step reasoning tasks because (1) DPO relies on paired preference data, which is not readily available for multi-step reasoning tasks, and (2) it treats all tokens uniformly, making it ineffective for credit assignment in multi-step reasoning tasks, which often come with sparse reward. In this w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16145","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16145","created_at":"2026-07-05T09:54:07.107806+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16145v2","created_at":"2026-07-05T09:54:07.107806+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16145","created_at":"2026-07-05T09:54:07.107806+00:00"},{"alias_kind":"pith_short_12","alias_value":"5RPKB5NPJNIZ","created_at":"2026-07-05T09:54:07.107806+00:00"},{"alias_kind":"pith_short_16","alias_value":"5RPKB5NPJNIZHS2Z","created_at":"2026-07-05T09:54:07.107806+00:00"},{"alias_kind":"pith_short_8","alias_value":"5RPKB5NP","created_at":"2026-07-05T09:54:07.107806+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":212,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":257,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":240,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08539","citing_title":"On the optimization dynamics of RLVR: Gradient gap and step size thresholds","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04937","citing_title":"Pramana: Fine-Tuning Large Language Models for Epistemic Reasoning through Navya-Nyaya","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ","json":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ.json","graph_json":"https://pith.science/api/pith-number/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/graph.json","events_json":"https://pith.science/api/pith-number/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/events.json","paper":"https://pith.science/paper/5RPKB5NP"},"agent_actions":{"view_html":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ","download_json":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ.json","view_paper":"https://pith.science/paper/5RPKB5NP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16145&json=true","fetch_graph":"https://pith.science/api/pith-number/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/graph.json","fetch_events":"https://pith.science/api/pith-number/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/action/storage_attestation","attest_author":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/action/author_attestation","sign_citation":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/action/citation_signature","submit_replication":"https://pith.science/pith/5RPKB5NPJNIZHS2ZZSMSNLE5JZ/action/replication_record"}},"created_at":"2026-07-05T09:54:07.107806+00:00","updated_at":"2026-07-05T09:54:07.107806+00:00"}