{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:S6UOY474D22ZXU4FNV54BR5YAD","short_pith_number":"pith:S6UOY474","canonical_record":{"source":{"id":"2505.18065","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T16:12:12Z","cross_cats_sorted":[],"title_canon_sha256":"197672b84fcbd168768a2555454f519bcbea4abb37c689e0c2bf67192724cd1a","abstract_canon_sha256":"59a03ce283bb9a5dbcda06192dfd701e268d419755166dd698b547ea37663f1a"},"schema_version":"1.0"},"canonical_sha256":"97a8ec73fc1eb59bd3856d7bc0c7b800cf29149bbc9322d8c0d0bbba7a2e6530","source":{"kind":"arxiv","id":"2505.18065","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.18065","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"arxiv_version","alias_value":"2505.18065v1","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18065","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"pith_short_12","alias_value":"S6UOY474D22Z","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"pith_short_16","alias_value":"S6UOY474D22ZXU4F","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"pith_short_8","alias_value":"S6UOY474","created_at":"2026-07-05T11:08:36Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:S6UOY474D22ZXU4FNV54BR5YAD","target":"record","payload":{"canonical_record":{"source":{"id":"2505.18065","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T16:12:12Z","cross_cats_sorted":[],"title_canon_sha256":"197672b84fcbd168768a2555454f519bcbea4abb37c689e0c2bf67192724cd1a","abstract_canon_sha256":"59a03ce283bb9a5dbcda06192dfd701e268d419755166dd698b547ea37663f1a"},"schema_version":"1.0"},"canonical_sha256":"97a8ec73fc1eb59bd3856d7bc0c7b800cf29149bbc9322d8c0d0bbba7a2e6530","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:36.167435Z","signature_b64":"15pqX4c73cg3kE9A+8OGMqkRiy3DFTyayMoLzRe78aGjJVxuhMrI/tvXuhSJR3+dOjP1GXdHgNBnGVbmrbY2BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"97a8ec73fc1eb59bd3856d7bc0c7b800cf29149bbc9322d8c0d0bbba7a2e6530","last_reissued_at":"2026-07-05T11:08:36.166989Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:36.166989Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.18065","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:08:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FSjY/FpFhJBlg255YlA+Fhm7cGunNsiSX5ba0ERrH5e+fku8A0PpCiP8BsAMxFlpvxK1bcYb4RLji6vj3E6jCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T18:03:53.407079Z"},"content_sha256":"304596945a2576daea56b2d47271b8ea114e1568195a38d6db2b01ada1e5e982","schema_version":"1.0","event_id":"sha256:304596945a2576daea56b2d47271b8ea114e1568195a38d6db2b01ada1e5e982"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:S6UOY474D22ZXU4FNV54BR5YAD","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reward Model Generalization for Compute-Aware Test-Time Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Changwen Zheng, Gang Hua, Siyu Zhao, Wenwen Qiang, Zeen Song","submitted_at":"2025-05-23T16:12:12Z","abstract_excerpt":"External test-time reasoning enhances large language models (LLMs) by decoupling generation and selection. At inference time, the model generates multiple reasoning paths, and an auxiliary process reward model (PRM) is used to score and select the best one. A central challenge in this setting is test-time compute optimality (TCO), i.e., how to maximize answer accuracy under a fixed inference budget. In this work, we establish a theoretical framework to analyze how the generalization error of the PRM affects compute efficiency and reasoning performance. Leveraging PAC-Bayes theory, we derive ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18065","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18065/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:08:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CBL2tMRpL5+zwjea9bhNwZCPkB0fYDX3w6U4C6en/k9MmMfsdQFhdPVDJd4RyGm1ALnaHDwImp5HE+sixto5Ag==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T18:03:53.407858Z"},"content_sha256":"1d4a3b71716f8ffe0fa364ac6ad045341fc2fbeef1b7bafb4191cb14c2a283c7","schema_version":"1.0","event_id":"sha256:1d4a3b71716f8ffe0fa364ac6ad045341fc2fbeef1b7bafb4191cb14c2a283c7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/bundle.json","state_url":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/S6UOY474D22ZXU4FNV54BR5YAD/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T18:03:53Z","links":{"resolver":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD","bundle":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/bundle.json","state":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/state.json","well_known_bundle":"https://pith.science/.well-known/pith/S6UOY474D22ZXU4FNV54BR5YAD/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:S6UOY474D22ZXU4FNV54BR5YAD","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"59a03ce283bb9a5dbcda06192dfd701e268d419755166dd698b547ea37663f1a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T16:12:12Z","title_canon_sha256":"197672b84fcbd168768a2555454f519bcbea4abb37c689e0c2bf67192724cd1a"},"schema_version":"1.0","source":{"id":"2505.18065","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.18065","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"arxiv_version","alias_value":"2505.18065v1","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18065","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"pith_short_12","alias_value":"S6UOY474D22Z","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"pith_short_16","alias_value":"S6UOY474D22ZXU4F","created_at":"2026-07-05T11:08:36Z"},{"alias_kind":"pith_short_8","alias_value":"S6UOY474","created_at":"2026-07-05T11:08:36Z"}],"graph_snapshots":[{"event_id":"sha256:1d4a3b71716f8ffe0fa364ac6ad045341fc2fbeef1b7bafb4191cb14c2a283c7","target":"graph","created_at":"2026-07-05T11:08:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.18065/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"External test-time reasoning enhances large language models (LLMs) by decoupling generation and selection. At inference time, the model generates multiple reasoning paths, and an auxiliary process reward model (PRM) is used to score and select the best one. A central challenge in this setting is test-time compute optimality (TCO), i.e., how to maximize answer accuracy under a fixed inference budget. In this work, we establish a theoretical framework to analyze how the generalization error of the PRM affects compute efficiency and reasoning performance. Leveraging PAC-Bayes theory, we derive ge","authors_text":"Changwen Zheng, Gang Hua, Siyu Zhao, Wenwen Qiang, Zeen Song","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T16:12:12Z","title":"Reward Model Generalization for Compute-Aware Test-Time Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18065","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:304596945a2576daea56b2d47271b8ea114e1568195a38d6db2b01ada1e5e982","target":"record","created_at":"2026-07-05T11:08:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"59a03ce283bb9a5dbcda06192dfd701e268d419755166dd698b547ea37663f1a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T16:12:12Z","title_canon_sha256":"197672b84fcbd168768a2555454f519bcbea4abb37c689e0c2bf67192724cd1a"},"schema_version":"1.0","source":{"id":"2505.18065","kind":"arxiv","version":1}},"canonical_sha256":"97a8ec73fc1eb59bd3856d7bc0c7b800cf29149bbc9322d8c0d0bbba7a2e6530","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"97a8ec73fc1eb59bd3856d7bc0c7b800cf29149bbc9322d8c0d0bbba7a2e6530","first_computed_at":"2026-07-05T11:08:36.166989Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:08:36.166989Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"15pqX4c73cg3kE9A+8OGMqkRiy3DFTyayMoLzRe78aGjJVxuhMrI/tvXuhSJR3+dOjP1GXdHgNBnGVbmrbY2BQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:08:36.167435Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.18065","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:304596945a2576daea56b2d47271b8ea114e1568195a38d6db2b01ada1e5e982","sha256:1d4a3b71716f8ffe0fa364ac6ad045341fc2fbeef1b7bafb4191cb14c2a283c7"],"state_sha256":"afa1307f950e78f92ac580b6da03aaf21b77325d1a7feddb2a25903df1654ab2"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dUpaGPtGs954NLY+p/A9gbbjchuzl9PdS6qggcYgeAXMKnO4GOy2PtQDE0Ht8ULpq7Un8qrJc7RxBqAhAxX0Ag==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T18:03:53.413028Z","bundle_sha256":"9a371fb8d6824a02204b91c2bcd895052de1923f81f50b2a07e3b6fca9ed4e0e"}}