{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:6XJTB6SV2QLJUQAGWOBIEEKDZC","short_pith_number":"pith:6XJTB6SV","canonical_record":{"source":{"id":"2506.08745","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"227090fd0cacc8c55d12f1bac32557e4e21fac1d4f7f02c6e773d082359e94b9","abstract_canon_sha256":"93ee5ad9ef88d09a9cf630e7ebf35d17781e25e542d73316f69e270228d07a1b"},"schema_version":"1.0"},"canonical_sha256":"f5d330fa55d4169a4006b382821143c887e98800a2a0b66f99a3ad96b4cc907f","source":{"kind":"arxiv","id":"2506.08745","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.08745","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"arxiv_version","alias_value":"2506.08745v1","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.08745","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_12","alias_value":"6XJTB6SV2QLJ","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_16","alias_value":"6XJTB6SV2QLJUQAG","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_8","alias_value":"6XJTB6SV","created_at":"2026-07-05T11:19:12Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:6XJTB6SV2QLJUQAGWOBIEEKDZC","target":"record","payload":{"canonical_record":{"source":{"id":"2506.08745","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"227090fd0cacc8c55d12f1bac32557e4e21fac1d4f7f02c6e773d082359e94b9","abstract_canon_sha256":"93ee5ad9ef88d09a9cf630e7ebf35d17781e25e542d73316f69e270228d07a1b"},"schema_version":"1.0"},"canonical_sha256":"f5d330fa55d4169a4006b382821143c887e98800a2a0b66f99a3ad96b4cc907f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:12.585029Z","signature_b64":"jXcdRres7OmDtqwx6brn2lt+M8r0em/5dEqreZUdPjxx0ebomo4MRUhghMin2keSKndXTOtrI+DIDwIHnfynBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5d330fa55d4169a4006b382821143c887e98800a2a0b66f99a3ad96b4cc907f","last_reissued_at":"2026-07-05T11:19:12.584543Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:12.584543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.08745","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:19:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JkWgmMY/UEcAisadyKP2Iilm7MlDMsZn8w6oXXsu8pOml2qCILdwn7Wjsq/9qg4jOCwrFVF+EYaPo7DMCwDuDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T13:26:59.979779Z"},"content_sha256":"40b4439aff83c31385cbfa9d43fcae62282b8613ab2f076ea539121013271d67","schema_version":"1.0","event_id":"sha256:40b4439aff83c31385cbfa9d43fcae62282b8613ab2f076ea539121013271d67"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:6XJTB6SV2QLJUQAGWOBIEEKDZC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Baisheng Lai, Dacheng Tao, Jieping Ye, Kongcheng Zhang, Mingli Song, Qi Yao, Shunyu Liu, Yingjie Wang","submitted_at":"2025-06-10T12:40:39Z","abstract_excerpt":"Recent advances of Reinforcement Learning (RL) have highlighted its potential in complex reasoning tasks, yet effective training often relies on external supervision, which limits the broader applicability. In this work, we propose a novel self-rewarding reinforcement learning framework to enhance Large Language Model (LLM) reasoning by leveraging the consistency of intermediate reasoning states across different reasoning trajectories. Our key insight is that correct responses often exhibit consistent trajectory patterns in terms of model likelihood: their intermediate reasoning states tend to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.08745","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.08745/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:19:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"LrkFo2JbySpa/PhkaAbjSLo8EUqKV0CWoUemt+Av5toFR1g0wfzWSLzcz7JfyA/g8+w2TFL6wBWbqKzFvSGOBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T13:26:59.980264Z"},"content_sha256":"7bb873a00786760d37629efbc3ef73370a7665fa0f52be407c5967b5b2e5325c","schema_version":"1.0","event_id":"sha256:7bb873a00786760d37629efbc3ef73370a7665fa0f52be407c5967b5b2e5325c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/6XJTB6SV2QLJUQAGWOBIEEKDZC/bundle.json","state_url":"https://pith.science/pith/6XJTB6SV2QLJUQAGWOBIEEKDZC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/6XJTB6SV2QLJUQAGWOBIEEKDZC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T13:26:59Z","links":{"resolver":"https://pith.science/pith/6XJTB6SV2QLJUQAGWOBIEEKDZC","bundle":"https://pith.science/pith/6XJTB6SV2QLJUQAGWOBIEEKDZC/bundle.json","state":"https://pith.science/pith/6XJTB6SV2QLJUQAGWOBIEEKDZC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/6XJTB6SV2QLJUQAGWOBIEEKDZC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:6XJTB6SV2QLJUQAGWOBIEEKDZC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"93ee5ad9ef88d09a9cf630e7ebf35d17781e25e542d73316f69e270228d07a1b","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","title_canon_sha256":"227090fd0cacc8c55d12f1bac32557e4e21fac1d4f7f02c6e773d082359e94b9"},"schema_version":"1.0","source":{"id":"2506.08745","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.08745","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"arxiv_version","alias_value":"2506.08745v1","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.08745","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_12","alias_value":"6XJTB6SV2QLJ","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_16","alias_value":"6XJTB6SV2QLJUQAG","created_at":"2026-07-05T11:19:12Z"},{"alias_kind":"pith_short_8","alias_value":"6XJTB6SV","created_at":"2026-07-05T11:19:12Z"}],"graph_snapshots":[{"event_id":"sha256:7bb873a00786760d37629efbc3ef73370a7665fa0f52be407c5967b5b2e5325c","target":"graph","created_at":"2026-07-05T11:19:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.08745/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advances of Reinforcement Learning (RL) have highlighted its potential in complex reasoning tasks, yet effective training often relies on external supervision, which limits the broader applicability. In this work, we propose a novel self-rewarding reinforcement learning framework to enhance Large Language Model (LLM) reasoning by leveraging the consistency of intermediate reasoning states across different reasoning trajectories. Our key insight is that correct responses often exhibit consistent trajectory patterns in terms of model likelihood: their intermediate reasoning states tend to","authors_text":"Baisheng Lai, Dacheng Tao, Jieping Ye, Kongcheng Zhang, Mingli Song, Qi Yao, Shunyu Liu, Yingjie Wang","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.08745","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:40b4439aff83c31385cbfa9d43fcae62282b8613ab2f076ea539121013271d67","target":"record","created_at":"2026-07-05T11:19:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"93ee5ad9ef88d09a9cf630e7ebf35d17781e25e542d73316f69e270228d07a1b","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-10T12:40:39Z","title_canon_sha256":"227090fd0cacc8c55d12f1bac32557e4e21fac1d4f7f02c6e773d082359e94b9"},"schema_version":"1.0","source":{"id":"2506.08745","kind":"arxiv","version":1}},"canonical_sha256":"f5d330fa55d4169a4006b382821143c887e98800a2a0b66f99a3ad96b4cc907f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f5d330fa55d4169a4006b382821143c887e98800a2a0b66f99a3ad96b4cc907f","first_computed_at":"2026-07-05T11:19:12.584543Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:19:12.584543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jXcdRres7OmDtqwx6brn2lt+M8r0em/5dEqreZUdPjxx0ebomo4MRUhghMin2keSKndXTOtrI+DIDwIHnfynBw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:19:12.585029Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.08745","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:40b4439aff83c31385cbfa9d43fcae62282b8613ab2f076ea539121013271d67","sha256:7bb873a00786760d37629efbc3ef73370a7665fa0f52be407c5967b5b2e5325c"],"state_sha256":"f7bd26c1483787f39d808220cfde6627b766f23bff9f599b8157e28c64d4b17f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1hxyY9ynqtGiPpt9ECBDzEEfapeopDt6nhKG+FmvJ7+LX1RuS5bCm2Y7I4r3rfPFqU7TvDLZhyPFCrxQHF0qAg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T13:26:59.983959Z","bundle_sha256":"ef704f2db153b98e160a53ba664a6f09a57cfc253e6784ddf462bc6d3ba0b90e"}}