{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:PRWL2ONAWFFRH5ONC4IH33PAMC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"622e28284020aaa3dfdddcccf494f26c7b16933ee401fb5c331a3dc8363713be","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T11:40:34Z","title_canon_sha256":"cc5740b79ead95572596d19cc0186c682dc6829a6d447c2febba90112812711e"},"schema_version":"1.0","source":{"id":"2505.23363","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.23363","created_at":"2026-07-05T11:12:00Z"},{"alias_kind":"arxiv_version","alias_value":"2505.23363v1","created_at":"2026-07-05T11:12:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23363","created_at":"2026-07-05T11:12:00Z"},{"alias_kind":"pith_short_12","alias_value":"PRWL2ONAWFFR","created_at":"2026-07-05T11:12:00Z"},{"alias_kind":"pith_short_16","alias_value":"PRWL2ONAWFFRH5ON","created_at":"2026-07-05T11:12:00Z"},{"alias_kind":"pith_short_8","alias_value":"PRWL2ONA","created_at":"2026-07-05T11:12:00Z"}],"graph_snapshots":[{"event_id":"sha256:4b2a1fa77e11ffe328aa0be36401d52a7d466ad6fc8b581257e481213744fa0f","target":"graph","created_at":"2026-07-05T11:12:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.23363/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Process reward models (PRMs) provide more nuanced supervision compared to outcome reward models (ORMs) for optimizing policy models, positioning them as a promising approach to enhancing the capabilities of LLMs in complex reasoning tasks. Recent efforts have advanced PRMs from step-level to token-level granularity by integrating reward modeling into the training of generative models, with reward scores derived from token generation probabilities. However, the conflict between generative language modeling and reward modeling may introduce instability and lead to inaccurate credit assignments. ","authors_text":"Hongtao Tian, Hongzhan Chen, Ruijun Chen, Shiping Gao, Tao Yang, Ting Yao, Xiaojun Quan","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T11:40:34Z","title":"Discriminative Policy Optimization for Token-Level Reward Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23363","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3ee7ca6fffd518b119e85a94af6c6caaaf70efd49a3dd675edfa5df7264fb67f","target":"record","created_at":"2026-07-05T11:12:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"622e28284020aaa3dfdddcccf494f26c7b16933ee401fb5c331a3dc8363713be","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T11:40:34Z","title_canon_sha256":"cc5740b79ead95572596d19cc0186c682dc6829a6d447c2febba90112812711e"},"schema_version":"1.0","source":{"id":"2505.23363","kind":"arxiv","version":1}},"canonical_sha256":"7c6cbd39a0b14b13f5cd17107dede060836a46846adaec6a4e18c25b6fd4f35e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7c6cbd39a0b14b13f5cd17107dede060836a46846adaec6a4e18c25b6fd4f35e","first_computed_at":"2026-07-05T11:12:00.328492Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:12:00.328492Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"76r/dyAL6abzbiHodOL8KAmDlcQo4xaR8kjO45HOeA2vvqbLnBT3ZhRW3PouzqtMeL+wILSm44JUcl3k6ckvDg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:12:00.328993Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.23363","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3ee7ca6fffd518b119e85a94af6c6caaaf70efd49a3dd675edfa5df7264fb67f","sha256:4b2a1fa77e11ffe328aa0be36401d52a7d466ad6fc8b581257e481213744fa0f"],"state_sha256":"90e2981d9c36097f960457749bfa648e4c58618924312b72b8c87207a4d27e79"}