{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:6TL5FWHPKL344KQTAA3PFAYFRW","short_pith_number":"pith:6TL5FWHP","canonical_record":{"source":{"id":"2410.09302","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-11T23:29:20Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"9bd45a3405b8be91c5cedfb568181bec387d2e776c34c2ad2b8c44171309521f","abstract_canon_sha256":"d8f20e6a6194aebd616ec1fc7961fe58d9cdd3c9098dde1ea7cca00c41e41107"},"schema_version":"1.0"},"canonical_sha256":"f4d7d2d8ef52f7ce2a130036f283058dba8b065b24852d78f84b3f6ee27fb8ac","source":{"kind":"arxiv","id":"2410.09302","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.09302","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"arxiv_version","alias_value":"2410.09302v2","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09302","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"pith_short_12","alias_value":"6TL5FWHPKL34","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"pith_short_16","alias_value":"6TL5FWHPKL344KQT","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"pith_short_8","alias_value":"6TL5FWHP","created_at":"2026-07-05T10:12:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:6TL5FWHPKL344KQTAA3PFAYFRW","target":"record","payload":{"canonical_record":{"source":{"id":"2410.09302","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-11T23:29:20Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"9bd45a3405b8be91c5cedfb568181bec387d2e776c34c2ad2b8c44171309521f","abstract_canon_sha256":"d8f20e6a6194aebd616ec1fc7961fe58d9cdd3c9098dde1ea7cca00c41e41107"},"schema_version":"1.0"},"canonical_sha256":"f4d7d2d8ef52f7ce2a130036f283058dba8b065b24852d78f84b3f6ee27fb8ac","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:21.209266Z","signature_b64":"Uj5aXJnzV+5jlAYWLItpgotr72bJKcvFhaTu91/r5a6x6Cy3izI2o2+xqDw9Wiq+XF3R6smvFzHZjIzSgJ1sAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4d7d2d8ef52f7ce2a130036f283058dba8b065b24852d78f84b3f6ee27fb8ac","last_reissued_at":"2026-07-05T10:12:21.208733Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:21.208733Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.09302","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:12:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fJjlz9yUGdi6IKseVaNzyCsjSAi8BCK+0H8DMRQ0EVd5L6PlkhFihzwbKfKcml+RjJuE5wCg9+7Ysbv+2di0AA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T12:32:01.045033Z"},"content_sha256":"f247de301057e2a80b469283e698b4c9634ee918dba8b3f4b4f0d0ca01cfe980","schema_version":"1.0","event_id":"sha256:f247de301057e2a80b469283e698b4c9634ee918dba8b3f4b4f0d0ca01cfe980"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:6TL5FWHPKL344KQTAA3PFAYFRW","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Enhancing Multi-Step Reasoning Abilities of Language Models through Direct Q-Function Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chen Dun, Guanlin Liu, Kaixuan Ji, Lin Yan, Ning Dai, Qingping Yang, Quanquan Gu, Renjie Zheng, Zheng Wu","submitted_at":"2024-10-11T23:29:20Z","abstract_excerpt":"Reinforcement Learning (RL) plays a crucial role in aligning large language models (LLMs) with human preferences and improving their ability to perform complex tasks. However, current approaches either require significant computational resources due to the use of multiple models and extensive online sampling for training (e.g., PPO) or are framed as bandit problems (e.g., DPO, DRO), which often struggle with multi-step reasoning tasks, such as math problem solving and complex reasoning that involve long chains of thought. To overcome these limitations, we introduce Direct Q-function Optimizati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09302","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:12:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PUUuQ73TJiUi9QUG2FQgj9THwO3aI/J1oBLeKdwP30ALI3pzbK5hf3fFlMz/ODtVfXQ8akJfKJbAtWMQGprcDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T12:32:01.045999Z"},"content_sha256":"e51966a25e8ecf08fa13731aa3e50bc3aad510386de9e8926503f61dbae18982","schema_version":"1.0","event_id":"sha256:e51966a25e8ecf08fa13731aa3e50bc3aad510386de9e8926503f61dbae18982"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/6TL5FWHPKL344KQTAA3PFAYFRW/bundle.json","state_url":"https://pith.science/pith/6TL5FWHPKL344KQTAA3PFAYFRW/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/6TL5FWHPKL344KQTAA3PFAYFRW/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T12:32:01Z","links":{"resolver":"https://pith.science/pith/6TL5FWHPKL344KQTAA3PFAYFRW","bundle":"https://pith.science/pith/6TL5FWHPKL344KQTAA3PFAYFRW/bundle.json","state":"https://pith.science/pith/6TL5FWHPKL344KQTAA3PFAYFRW/state.json","well_known_bundle":"https://pith.science/.well-known/pith/6TL5FWHPKL344KQTAA3PFAYFRW/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:6TL5FWHPKL344KQTAA3PFAYFRW","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d8f20e6a6194aebd616ec1fc7961fe58d9cdd3c9098dde1ea7cca00c41e41107","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-11T23:29:20Z","title_canon_sha256":"9bd45a3405b8be91c5cedfb568181bec387d2e776c34c2ad2b8c44171309521f"},"schema_version":"1.0","source":{"id":"2410.09302","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.09302","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"arxiv_version","alias_value":"2410.09302v2","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09302","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"pith_short_12","alias_value":"6TL5FWHPKL34","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"pith_short_16","alias_value":"6TL5FWHPKL344KQT","created_at":"2026-07-05T10:12:21Z"},{"alias_kind":"pith_short_8","alias_value":"6TL5FWHP","created_at":"2026-07-05T10:12:21Z"}],"graph_snapshots":[{"event_id":"sha256:e51966a25e8ecf08fa13731aa3e50bc3aad510386de9e8926503f61dbae18982","target":"graph","created_at":"2026-07-05T10:12:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.09302/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning (RL) plays a crucial role in aligning large language models (LLMs) with human preferences and improving their ability to perform complex tasks. However, current approaches either require significant computational resources due to the use of multiple models and extensive online sampling for training (e.g., PPO) or are framed as bandit problems (e.g., DPO, DRO), which often struggle with multi-step reasoning tasks, such as math problem solving and complex reasoning that involve long chains of thought. To overcome these limitations, we introduce Direct Q-function Optimizati","authors_text":"Chen Dun, Guanlin Liu, Kaixuan Ji, Lin Yan, Ning Dai, Qingping Yang, Quanquan Gu, Renjie Zheng, Zheng Wu","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-11T23:29:20Z","title":"Enhancing Multi-Step Reasoning Abilities of Language Models through Direct Q-Function Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09302","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f247de301057e2a80b469283e698b4c9634ee918dba8b3f4b4f0d0ca01cfe980","target":"record","created_at":"2026-07-05T10:12:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d8f20e6a6194aebd616ec1fc7961fe58d9cdd3c9098dde1ea7cca00c41e41107","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-11T23:29:20Z","title_canon_sha256":"9bd45a3405b8be91c5cedfb568181bec387d2e776c34c2ad2b8c44171309521f"},"schema_version":"1.0","source":{"id":"2410.09302","kind":"arxiv","version":2}},"canonical_sha256":"f4d7d2d8ef52f7ce2a130036f283058dba8b065b24852d78f84b3f6ee27fb8ac","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f4d7d2d8ef52f7ce2a130036f283058dba8b065b24852d78f84b3f6ee27fb8ac","first_computed_at":"2026-07-05T10:12:21.208733Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:12:21.208733Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Uj5aXJnzV+5jlAYWLItpgotr72bJKcvFhaTu91/r5a6x6Cy3izI2o2+xqDw9Wiq+XF3R6smvFzHZjIzSgJ1sAg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:12:21.209266Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.09302","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f247de301057e2a80b469283e698b4c9634ee918dba8b3f4b4f0d0ca01cfe980","sha256:e51966a25e8ecf08fa13731aa3e50bc3aad510386de9e8926503f61dbae18982"],"state_sha256":"b9fee50b449136a1259957a75d3206cf28d2d8bee10f2c682d7bfabcfff7511c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"b2K8BPAz+OwAir5UlKK7murHgkf8mrPLtOIHpsyG5+7ccOZfA6SZldBf24RHF2ih5cqvPrmKK5nHwQsuq+78Cg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T12:32:01.051630Z","bundle_sha256":"45f1a35bc5898e27e234123a51eac34144cf2d72d9daaad5f595b446d0225695"}}