{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:6XJTB6SV2QLJUQAGWOBIEEKDZC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"93ee5ad9ef88d09a9cf630e7ebf35d17781e25e542d73316f69e270228d07a1b","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","title_canon_sha256":"227090fd0cacc8c55d12f1bac32557e4e21fac1d4f7f02c6e773d082359e94b9"},"schema_version":"1.0","source":{"id":"2506.08745","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.08745","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"arxiv_version","alias_value":"2506.08745v1","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.08745","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_12","alias_value":"6XJTB6SV2QLJ","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_16","alias_value":"6XJTB6SV2QLJUQAG","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_8","alias_value":"6XJTB6SV","created_at":"2026-07-05T11:19:12Z"}],"graph_snapshots":[{"event_id":"sha256:7bb873a00786760d37629efbc3ef73370a7665fa0f52be407c5967b5b2e5325c","target":"graph","created_at":"2026-07-05T11:19:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.08745/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advances of Reinforcement Learning (RL) have highlighted its potential in complex reasoning tasks, yet effective training often relies on external supervision, which limits the broader applicability. In this work, we propose a novel self-rewarding reinforcement learning framework to enhance Large Language Model (LLM) reasoning by leveraging the consistency of intermediate reasoning states across different reasoning trajectories. Our key insight is that correct responses often exhibit consistent trajectory patterns in terms of model likelihood: their intermediate reasoning states tend to","authors_text":"Baisheng Lai, Dacheng Tao, Jieping Ye, Kongcheng Zhang, Mingli Song, Qi Yao, Shunyu Liu, Yingjie Wang","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.08745","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:40b4439aff83c31385cbfa9d43fcae62282b8613ab2f076ea539121013271d67","target":"record","created_at":"2026-07-05T11:19:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"93ee5ad9ef88d09a9cf630e7ebf35d17781e25e542d73316f69e270228d07a1b","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","title_canon_sha256":"227090fd0cacc8c55d12f1bac32557e4e21fac1d4f7f02c6e773d082359e94b9"},"schema_version":"1.0","source":{"id":"2506.08745","kind":"arxiv","version":1}},"canonical_sha256":"f5d330fa55d4169a4006b382821143c887e98800a2a0b66f99a3ad96b4cc907f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f5d330fa55d4169a4006b382821143c887e98800a2a0b66f99a3ad96b4cc907f","first_computed_at":"2026-07-05T11:19:12.584543Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:19:12.584543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jXcdRres7OmDtqwx6brn2lt+M8r0em/5dEqreZUdPjxx0ebomo4MRUhghMin2keSKndXTOtrI+DIDwIHnfynBw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:19:12.585029Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.08745","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:40b4439aff83c31385cbfa9d43fcae62282b8613ab2f076ea539121013271d67","sha256:7bb873a00786760d37629efbc3ef73370a7665fa0f52be407c5967b5b2e5325c"],"state_sha256":"f7bd26c1483787f39d808220cfde6627b766f23bff9f599b8157e28c64d4b17f"}