{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:B76QZQP7MXSOPZJ7ICV2DHWWLQ","short_pith_number":"pith:B76QZQP7","canonical_record":{"source":{"id":"2505.16022","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T21:12:35Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3137b468f2eda40bf3ebeb1c019ac77165cfaaca98aeb68103f67e37cda2e9cd","abstract_canon_sha256":"f851aaf8425f9b90ee3f1e9aee0a5c359adbb76d48f39ab1f35606b7dfbad819"},"schema_version":"1.0"},"canonical_sha256":"0ffd0cc1ff65e4e7e53f40aba19ed65c09fe5eac13a5e42b505a0c9628b0c337","source":{"kind":"arxiv","id":"2505.16022","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.16022","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"arxiv_version","alias_value":"2505.16022v2","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16022","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"pith_short_12","alias_value":"B76QZQP7MXSO","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"pith_short_16","alias_value":"B76QZQP7MXSOPZJ7","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"pith_short_8","alias_value":"B76QZQP7","created_at":"2026-07-05T12:03:48Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:B76QZQP7MXSOPZJ7ICV2DHWWLQ","target":"record","payload":{"canonical_record":{"source":{"id":"2505.16022","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T21:12:35Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3137b468f2eda40bf3ebeb1c019ac77165cfaaca98aeb68103f67e37cda2e9cd","abstract_canon_sha256":"f851aaf8425f9b90ee3f1e9aee0a5c359adbb76d48f39ab1f35606b7dfbad819"},"schema_version":"1.0"},"canonical_sha256":"0ffd0cc1ff65e4e7e53f40aba19ed65c09fe5eac13a5e42b505a0c9628b0c337","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:48.659415Z","signature_b64":"VQjkoPhT2Gg5pmSn2QVFL69MT+p2UHseVopnNWuJFx1U9SCBfoeSs5Y1Ixpj5M5deF0s6M3VNHSG2KxreF5mDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ffd0cc1ff65e4e7e53f40aba19ed65c09fe5eac13a5e42b505a0c9628b0c337","last_reissued_at":"2026-07-05T12:03:48.658882Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:48.658882Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.16022","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:03:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mZggcbbxllJzYfZsVDdo3tXORQi0Ikag1XyaEEACbZkh1CElPonQdOVMRZ/KiuYahFRew0M/lWD7OgEYQwMAAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T22:46:10.285661Z"},"content_sha256":"e0d8baeeab170d14749bbcd1b85f2bfcdf7eba45a0192755811c2abae9104719","schema_version":"1.0","event_id":"sha256:e0d8baeeab170d14749bbcd1b85f2bfcdf7eba45a0192755811c2abae9104719"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:B76QZQP7MXSOPZJ7ICV2DHWWLQ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"NOVER: Incentive Training for Language Models via Verifier-Free Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chen Qian, Siya Qi, Wei Liu, Xinyu Wang, Yali Du, Yulan He","submitted_at":"2025-05-21T21:12:35Z","abstract_excerpt":"Recent advances such as DeepSeek R1-Zero highlight the effectiveness of incentive training, a reinforcement learning paradigm that computes rewards solely based on the final answer part of a language model's output, thereby encouraging the generation of intermediate reasoning steps. However, these methods fundamentally rely on external verifiers, which limits their applicability to domains like mathematics and coding where such verifiers are readily available. Although reward models can serve as verifiers, they require high-quality annotated data and are costly to train. In this work, we propo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16022","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16022/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:03:48Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"f9mtg9xDA6MK6XtMSTytoG8r1nId7/220rmCmdog4q8RkoHtH9Fy+ctfwH8wbvvjScy1nSAPdUhIwaTfHSeABQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T22:46:10.286145Z"},"content_sha256":"1ec0a08b823ef997bfd6bcc083a636a9f73b2a69d03566761ccc005aef3b502b","schema_version":"1.0","event_id":"sha256:1ec0a08b823ef997bfd6bcc083a636a9f73b2a69d03566761ccc005aef3b502b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/B76QZQP7MXSOPZJ7ICV2DHWWLQ/bundle.json","state_url":"https://pith.science/pith/B76QZQP7MXSOPZJ7ICV2DHWWLQ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/B76QZQP7MXSOPZJ7ICV2DHWWLQ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T22:46:10Z","links":{"resolver":"https://pith.science/pith/B76QZQP7MXSOPZJ7ICV2DHWWLQ","bundle":"https://pith.science/pith/B76QZQP7MXSOPZJ7ICV2DHWWLQ/bundle.json","state":"https://pith.science/pith/B76QZQP7MXSOPZJ7ICV2DHWWLQ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/B76QZQP7MXSOPZJ7ICV2DHWWLQ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:B76QZQP7MXSOPZJ7ICV2DHWWLQ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f851aaf8425f9b90ee3f1e9aee0a5c359adbb76d48f39ab1f35606b7dfbad819","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T21:12:35Z","title_canon_sha256":"3137b468f2eda40bf3ebeb1c019ac77165cfaaca98aeb68103f67e37cda2e9cd"},"schema_version":"1.0","source":{"id":"2505.16022","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.16022","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"arxiv_version","alias_value":"2505.16022v2","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16022","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"pith_short_12","alias_value":"B76QZQP7MXSO","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"pith_short_16","alias_value":"B76QZQP7MXSOPZJ7","created_at":"2026-07-05T12:03:48Z"},{"alias_kind":"pith_short_8","alias_value":"B76QZQP7","created_at":"2026-07-05T12:03:48Z"}],"graph_snapshots":[{"event_id":"sha256:1ec0a08b823ef997bfd6bcc083a636a9f73b2a69d03566761ccc005aef3b502b","target":"graph","created_at":"2026-07-05T12:03:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.16022/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advances such as DeepSeek R1-Zero highlight the effectiveness of incentive training, a reinforcement learning paradigm that computes rewards solely based on the final answer part of a language model's output, thereby encouraging the generation of intermediate reasoning steps. However, these methods fundamentally rely on external verifiers, which limits their applicability to domains like mathematics and coding where such verifiers are readily available. Although reward models can serve as verifiers, they require high-quality annotated data and are costly to train. In this work, we propo","authors_text":"Chen Qian, Siya Qi, Wei Liu, Xinyu Wang, Yali Du, Yulan He","cross_cats":["cs.AI","cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T21:12:35Z","title":"NOVER: Incentive Training for Language Models via Verifier-Free Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16022","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:e0d8baeeab170d14749bbcd1b85f2bfcdf7eba45a0192755811c2abae9104719","target":"record","created_at":"2026-07-05T12:03:48Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f851aaf8425f9b90ee3f1e9aee0a5c359adbb76d48f39ab1f35606b7dfbad819","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T21:12:35Z","title_canon_sha256":"3137b468f2eda40bf3ebeb1c019ac77165cfaaca98aeb68103f67e37cda2e9cd"},"schema_version":"1.0","source":{"id":"2505.16022","kind":"arxiv","version":2}},"canonical_sha256":"0ffd0cc1ff65e4e7e53f40aba19ed65c09fe5eac13a5e42b505a0c9628b0c337","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0ffd0cc1ff65e4e7e53f40aba19ed65c09fe5eac13a5e42b505a0c9628b0c337","first_computed_at":"2026-07-05T12:03:48.658882Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:03:48.658882Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"VQjkoPhT2Gg5pmSn2QVFL69MT+p2UHseVopnNWuJFx1U9SCBfoeSs5Y1Ixpj5M5deF0s6M3VNHSG2KxreF5mDw==","signature_status":"signed_v1","signed_at":"2026-07-05T12:03:48.659415Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.16022","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:e0d8baeeab170d14749bbcd1b85f2bfcdf7eba45a0192755811c2abae9104719","sha256:1ec0a08b823ef997bfd6bcc083a636a9f73b2a69d03566761ccc005aef3b502b"],"state_sha256":"3559892203c3978d7f5900576ab0e49759551866d22cf0e136be75a2cdbfeaa8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gDR8Cnc8UQSH+m/82cKB1AijsddwW6PXOSdktQIjfhUXh3gVC/2n4zDWgsPQd6cer3FUrbmrjLIrbMyKvwjGAg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T22:46:10.289375Z","bundle_sha256":"75d5445c48ecc00215fb9d3ce3f01f4d17a56a4a6e0e6a6e2b88f00747657b1d"}}