{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:4EFXR5UNO7ZGMZ5QBVSS4G4AXT","short_pith_number":"pith:4EFXR5UN","canonical_record":{"source":{"id":"2506.16712","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-20T03:10:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1673e4df5b965eca21f00ced806160b573a0d6a0bbae495a74438b65ccb49fd6","abstract_canon_sha256":"f77bc0a226acffa880c15dca117c2c481ae3ad88be08637c01ab76e29ecb6259"},"schema_version":"1.0"},"canonical_sha256":"e10b78f68d77f26667b00d652e1b80bccf01dabfa7ed71567989c1cf5844b23f","source":{"kind":"arxiv","id":"2506.16712","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.16712","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"arxiv_version","alias_value":"2506.16712v1","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.16712","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"pith_short_12","alias_value":"4EFXR5UNO7ZG","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"pith_short_16","alias_value":"4EFXR5UNO7ZGMZ5Q","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"pith_short_8","alias_value":"4EFXR5UN","created_at":"2026-07-05T11:24:41Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:4EFXR5UNO7ZGMZ5QBVSS4G4AXT","target":"record","payload":{"canonical_record":{"source":{"id":"2506.16712","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-20T03:10:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1673e4df5b965eca21f00ced806160b573a0d6a0bbae495a74438b65ccb49fd6","abstract_canon_sha256":"f77bc0a226acffa880c15dca117c2c481ae3ad88be08637c01ab76e29ecb6259"},"schema_version":"1.0"},"canonical_sha256":"e10b78f68d77f26667b00d652e1b80bccf01dabfa7ed71567989c1cf5844b23f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:41.470994Z","signature_b64":"96qI5bjQhyHAPyN6IFLMlXeNSH6O4qpr1/2TRgvEGVLRX22GIufTdSdiTRqWa7EUkvNFaL8cBtdy2B2fp/JaDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e10b78f68d77f26667b00d652e1b80bccf01dabfa7ed71567989c1cf5844b23f","last_reissued_at":"2026-07-05T11:24:41.470459Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:41.470459Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.16712","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:24:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ORec4vO2k0UnygJyYnsP8KzGDrHeMcTgtl9cNHjsv3unHpIq2CgqN4lhxKl+x9rvCGPk4YC/bQeAt8ifD5vZDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T13:02:13.514370Z"},"content_sha256":"88b82425188444841742a7ece6ae5ca6f8033e1b6e61a5b645a2d3eef4c92aa8","schema_version":"1.0","event_id":"sha256:88b82425188444841742a7ece6ae5ca6f8033e1b6e61a5b645a2d3eef4c92aa8"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:4EFXR5UNO7ZGMZ5QBVSS4G4AXT","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"ReasonGRM: Enhancing Generative Reward Models through Large Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bin Chen, Bing-Kun Bao, Chuanrui Hu, Hua Zhang, Penghang Yu, Xinzge Gao","submitted_at":"2025-06-20T03:10:52Z","abstract_excerpt":"Generative Reward Models (GRMs) provide greater flexibility than scalar reward models in capturing human preferences, but their effectiveness is limited by poor reasoning capabilities. This often results in incomplete or overly speculative reasoning paths, leading to hallucinations or missing key information in complex tasks. We address this challenge with ReasonGRM, a three-stage generative reward modeling framework. In the first stage, Zero-RL is used to generate concise, outcome-directed reasoning paths that reduce the likelihood of critical omissions. In the second stage, we introduce a no"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.16712","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.16712/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:24:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uFukCiD5knyCtDArql5XZJt4BUq17connPti/iviH1Mg6IJLZ1DkhegNca4oxNfqo7clWXR44DavZtgt12bKAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T13:02:13.514930Z"},"content_sha256":"5a500193f748fcdf873d23572a372b0b65c210e0b789def1bdb3e0f909950971","schema_version":"1.0","event_id":"sha256:5a500193f748fcdf873d23572a372b0b65c210e0b789def1bdb3e0f909950971"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4EFXR5UNO7ZGMZ5QBVSS4G4AXT/bundle.json","state_url":"https://pith.science/pith/4EFXR5UNO7ZGMZ5QBVSS4G4AXT/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4EFXR5UNO7ZGMZ5QBVSS4G4AXT/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T13:02:13Z","links":{"resolver":"https://pith.science/pith/4EFXR5UNO7ZGMZ5QBVSS4G4AXT","bundle":"https://pith.science/pith/4EFXR5UNO7ZGMZ5QBVSS4G4AXT/bundle.json","state":"https://pith.science/pith/4EFXR5UNO7ZGMZ5QBVSS4G4AXT/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4EFXR5UNO7ZGMZ5QBVSS4G4AXT/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:4EFXR5UNO7ZGMZ5QBVSS4G4AXT","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f77bc0a226acffa880c15dca117c2c481ae3ad88be08637c01ab76e29ecb6259","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-20T03:10:52Z","title_canon_sha256":"1673e4df5b965eca21f00ced806160b573a0d6a0bbae495a74438b65ccb49fd6"},"schema_version":"1.0","source":{"id":"2506.16712","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.16712","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"arxiv_version","alias_value":"2506.16712v1","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.16712","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"pith_short_12","alias_value":"4EFXR5UNO7ZG","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"pith_short_16","alias_value":"4EFXR5UNO7ZGMZ5Q","created_at":"2026-07-05T11:24:41Z"},{"alias_kind":"pith_short_8","alias_value":"4EFXR5UN","created_at":"2026-07-05T11:24:41Z"}],"graph_snapshots":[{"event_id":"sha256:5a500193f748fcdf873d23572a372b0b65c210e0b789def1bdb3e0f909950971","target":"graph","created_at":"2026-07-05T11:24:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.16712/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Generative Reward Models (GRMs) provide greater flexibility than scalar reward models in capturing human preferences, but their effectiveness is limited by poor reasoning capabilities. This often results in incomplete or overly speculative reasoning paths, leading to hallucinations or missing key information in complex tasks. We address this challenge with ReasonGRM, a three-stage generative reward modeling framework. In the first stage, Zero-RL is used to generate concise, outcome-directed reasoning paths that reduce the likelihood of critical omissions. In the second stage, we introduce a no","authors_text":"Bin Chen, Bing-Kun Bao, Chuanrui Hu, Hua Zhang, Penghang Yu, Xinzge Gao","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-20T03:10:52Z","title":"ReasonGRM: Enhancing Generative Reward Models through Large Reasoning Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.16712","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:88b82425188444841742a7ece6ae5ca6f8033e1b6e61a5b645a2d3eef4c92aa8","target":"record","created_at":"2026-07-05T11:24:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f77bc0a226acffa880c15dca117c2c481ae3ad88be08637c01ab76e29ecb6259","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-20T03:10:52Z","title_canon_sha256":"1673e4df5b965eca21f00ced806160b573a0d6a0bbae495a74438b65ccb49fd6"},"schema_version":"1.0","source":{"id":"2506.16712","kind":"arxiv","version":1}},"canonical_sha256":"e10b78f68d77f26667b00d652e1b80bccf01dabfa7ed71567989c1cf5844b23f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e10b78f68d77f26667b00d652e1b80bccf01dabfa7ed71567989c1cf5844b23f","first_computed_at":"2026-07-05T11:24:41.470459Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:24:41.470459Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"96qI5bjQhyHAPyN6IFLMlXeNSH6O4qpr1/2TRgvEGVLRX22GIufTdSdiTRqWa7EUkvNFaL8cBtdy2B2fp/JaDA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:24:41.470994Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.16712","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:88b82425188444841742a7ece6ae5ca6f8033e1b6e61a5b645a2d3eef4c92aa8","sha256:5a500193f748fcdf873d23572a372b0b65c210e0b789def1bdb3e0f909950971"],"state_sha256":"b5297a247cf5ec898347c06e3d83c7e5de7cf3a334b355dd2ec26cf52b1d2e7e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xseT3Hy1PFnqepc+O2mq6gqkTSW0jCld8cFdixi9tduqoIWVWwltKYW6gvTIL2ODn0FCcRAvuzQhnXI3GQEVCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T13:02:13.519375Z","bundle_sha256":"45649bd9f0ceee220935278a15d8fc6ce280d84dd9143d9cd764449b4dab7fa7"}}