{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:S6UOY474D22ZXU4FNV54BR5YAD","short_pith_number":"pith:S6UOY474","schema_version":"1.0","canonical_sha256":"97a8ec73fc1eb59bd3856d7bc0c7b800cf29149bbc9322d8c0d0bbba7a2e6530","source":{"kind":"arxiv","id":"2505.18065","version":1},"attestation_state":"computed","paper":{"title":"Reward Model Generalization for Compute-Aware Test-Time Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Changwen Zheng, Gang Hua, Siyu Zhao, Wenwen Qiang, Zeen Song","submitted_at":"2025-05-23T16:12:12Z","abstract_excerpt":"External test-time reasoning enhances large language models (LLMs) by decoupling generation and selection. At inference time, the model generates multiple reasoning paths, and an auxiliary process reward model (PRM) is used to score and select the best one. A central challenge in this setting is test-time compute optimality (TCO), i.e., how to maximize answer accuracy under a fixed inference budget. In this work, we establish a theoretical framework to analyze how the generalization error of the PRM affects compute efficiency and reasoning performance. Leveraging PAC-Bayes theory, we derive ge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18065","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T16:12:12Z","cross_cats_sorted":[],"title_canon_sha256":"197672b84fcbd168768a2555454f519bcbea4abb37c689e0c2bf67192724cd1a","abstract_canon_sha256":"59a03ce283bb9a5dbcda06192dfd701e268d419755166dd698b547ea37663f1a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:36.167435Z","signature_b64":"15pqX4c73cg3kE9A+8OGMqkRiy3DFTyayMoLzRe78aGjJVxuhMrI/tvXuhSJR3+dOjP1GXdHgNBnGVbmrbY2BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"97a8ec73fc1eb59bd3856d7bc0c7b800cf29149bbc9322d8c0d0bbba7a2e6530","last_reissued_at":"2026-07-05T11:08:36.166989Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:36.166989Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward Model Generalization for Compute-Aware Test-Time Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Changwen Zheng, Gang Hua, Siyu Zhao, Wenwen Qiang, Zeen Song","submitted_at":"2025-05-23T16:12:12Z","abstract_excerpt":"External test-time reasoning enhances large language models (LLMs) by decoupling generation and selection. At inference time, the model generates multiple reasoning paths, and an auxiliary process reward model (PRM) is used to score and select the best one. A central challenge in this setting is test-time compute optimality (TCO), i.e., how to maximize answer accuracy under a fixed inference budget. In this work, we establish a theoretical framework to analyze how the generalization error of the PRM affects compute efficiency and reasoning performance. Leveraging PAC-Bayes theory, we derive ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18065","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18065/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18065","created_at":"2026-07-05T11:08:36.167046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18065v1","created_at":"2026-07-05T11:08:36.167046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18065","created_at":"2026-07-05T11:08:36.167046+00:00"},{"alias_kind":"pith_short_12","alias_value":"S6UOY474D22Z","created_at":"2026-07-05T11:08:36.167046+00:00"},{"alias_kind":"pith_short_16","alias_value":"S6UOY474D22ZXU4F","created_at":"2026-07-05T11:08:36.167046+00:00"},{"alias_kind":"pith_short_8","alias_value":"S6UOY474","created_at":"2026-07-05T11:08:36.167046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD","json":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD.json","graph_json":"https://pith.science/api/pith-number/S6UOY474D22ZXU4FNV54BR5YAD/graph.json","events_json":"https://pith.science/api/pith-number/S6UOY474D22ZXU4FNV54BR5YAD/events.json","paper":"https://pith.science/paper/S6UOY474"},"agent_actions":{"view_html":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD","download_json":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD.json","view_paper":"https://pith.science/paper/S6UOY474","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18065&json=true","fetch_graph":"https://pith.science/api/pith-number/S6UOY474D22ZXU4FNV54BR5YAD/graph.json","fetch_events":"https://pith.science/api/pith-number/S6UOY474D22ZXU4FNV54BR5YAD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/action/storage_attestation","attest_author":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/action/author_attestation","sign_citation":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/action/citation_signature","submit_replication":"https://pith.science/pith/S6UOY474D22ZXU4FNV54BR5YAD/action/replication_record"}},"created_at":"2026-07-05T11:08:36.167046+00:00","updated_at":"2026-07-05T11:08:36.167046+00:00"}