{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MWPBOOMLK3XGTVHDGHEQFCVGDN","short_pith_number":"pith:MWPBOOML","schema_version":"1.0","canonical_sha256":"659e17398b56ee69d4e331c9028aa61b44c91114ada8428da7f54f6cf5892aa1","source":{"kind":"arxiv","id":"2510.00492","version":3},"attestation_state":"computed","paper":{"title":"Rethinking Reward Models for Multi-Domain Test-Time Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dominik Wagner, Dong Bok Lee, Dongki Kim, Heejun Lee, Jiang Bian, Jingjing Fu, Jinheon Baek, Jinyu Wang, Jiongdao Jin, Lei Song, Minki Kang, Sangwoo Park, Seanie Lee, Sung Ju Hwang, Tobias Bocklet","submitted_at":"2025-10-01T04:21:14Z","abstract_excerpt":"The reliability of large language models (LLMs) during test-time scaling is often assessed with \\emph{external verifiers} or \\emph{reward models} that distinguish correct reasoning from flawed logic. Prior work has studied both outcome reward models (ORMs), which assess only the final answer, and process reward models (PRMs), which score intermediate reasoning steps. Although PRMs are often viewed as advantageous due to their finer-grained supervision, much of the supporting evidence comes from math-adjacent settings, and their relative benefits across broader domains remain unclear. We presen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.00492","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-10-01T04:21:14Z","cross_cats_sorted":[],"title_canon_sha256":"09729f63a62eaeea58d269bc27b03deed6efd725bae462158451ed900da57e5d","abstract_canon_sha256":"496b494e03389a1e856128de68e1297249a275a3d83478bfe20f8653d2d53074"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T00:21:12.982625Z","signature_b64":"2YCS+N1RxOQEtwrFxOebCnfWnc0fJ0KkorR1jftO37Igztr/HBGNtyokoKYsnctRVus+cOjigP414Rb4AMfkAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"659e17398b56ee69d4e331c9028aa61b44c91114ada8428da7f54f6cf5892aa1","last_reissued_at":"2026-07-15T00:21:12.981700Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T00:21:12.981700Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking Reward Models for Multi-Domain Test-Time Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dominik Wagner, Dong Bok Lee, Dongki Kim, Heejun Lee, Jiang Bian, Jingjing Fu, Jinheon Baek, Jinyu Wang, Jiongdao Jin, Lei Song, Minki Kang, Sangwoo Park, Seanie Lee, Sung Ju Hwang, Tobias Bocklet","submitted_at":"2025-10-01T04:21:14Z","abstract_excerpt":"The reliability of large language models (LLMs) during test-time scaling is often assessed with \\emph{external verifiers} or \\emph{reward models} that distinguish correct reasoning from flawed logic. Prior work has studied both outcome reward models (ORMs), which assess only the final answer, and process reward models (PRMs), which score intermediate reasoning steps. Although PRMs are often viewed as advantageous due to their finer-grained supervision, much of the supporting evidence comes from math-adjacent settings, and their relative benefits across broader domains remain unclear. We presen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.00492","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.00492/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.00492","created_at":"2026-07-15T00:21:12.982129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.00492v3","created_at":"2026-07-15T00:21:12.982129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.00492","created_at":"2026-07-15T00:21:12.982129+00:00"},{"alias_kind":"pith_short_12","alias_value":"MWPBOOMLK3XG","created_at":"2026-07-15T00:21:12.982129+00:00"},{"alias_kind":"pith_short_16","alias_value":"MWPBOOMLK3XGTVHD","created_at":"2026-07-15T00:21:12.982129+00:00"},{"alias_kind":"pith_short_8","alias_value":"MWPBOOML","created_at":"2026-07-15T00:21:12.982129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.00564","citing_title":"Decomposed On-Policy Distillation for Vision-Language Reasoning: Steering Gradients for Visual Grounding","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2605.13467","citing_title":"PDCR: Perception-Decomposed Confidence Reward for Vision-Language Reasoning","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN","json":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN.json","graph_json":"https://pith.science/api/pith-number/MWPBOOMLK3XGTVHDGHEQFCVGDN/graph.json","events_json":"https://pith.science/api/pith-number/MWPBOOMLK3XGTVHDGHEQFCVGDN/events.json","paper":"https://pith.science/paper/MWPBOOML"},"agent_actions":{"view_html":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN","download_json":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN.json","view_paper":"https://pith.science/paper/MWPBOOML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.00492&json=true","fetch_graph":"https://pith.science/api/pith-number/MWPBOOMLK3XGTVHDGHEQFCVGDN/graph.json","fetch_events":"https://pith.science/api/pith-number/MWPBOOMLK3XGTVHDGHEQFCVGDN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN/action/storage_attestation","attest_author":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN/action/author_attestation","sign_citation":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN/action/citation_signature","submit_replication":"https://pith.science/pith/MWPBOOMLK3XGTVHDGHEQFCVGDN/action/replication_record"}},"created_at":"2026-07-15T00:21:12.982129+00:00","updated_at":"2026-07-15T00:21:12.982129+00:00"}