{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DLM7TVE7P3X6Z6JM7BEHV6LZIN","short_pith_number":"pith:DLM7TVE7","schema_version":"1.0","canonical_sha256":"1ad9f9d49f7eefecf92cf8487af979437be51b33fd39e77f1fa2a467a6d053e4","source":{"kind":"arxiv","id":"2503.21295","version":1},"attestation_state":"computed","paper":{"title":"R-PRM: Reasoning-Driven Process Reward Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiajun Chen, Junxiao Liu, Shuaijie She, Shujian Huang, Xin Huang, Yifeng Liu","submitted_at":"2025-03-27T09:23:08Z","abstract_excerpt":"Large language models (LLMs) inevitably make mistakes when performing step-by-step mathematical reasoning. Process Reward Models (PRMs) have emerged as a promising solution by evaluating each reasoning step. However, existing PRMs typically output evaluation scores directly, limiting both learning efficiency and evaluation accuracy, which is further exacerbated by the scarcity of annotated data. To address these issues, we propose Reasoning-Driven Process Reward Modeling (R-PRM). First, we leverage stronger LLMs to generate seed data from limited annotations, effectively bootstrapping our mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.21295","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-03-27T09:23:08Z","cross_cats_sorted":[],"title_canon_sha256":"820083602dbad970920a2ad2bd86a7fd9d1359a2e325df242680fa7b9e36d6d5","abstract_canon_sha256":"7b89c429c7925abb1044b41fca21a91c865d6dad7a66f87346b13e6698b1a59e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:22.669532Z","signature_b64":"cc5+GVSDRYdLBM9I+pMOVNUrO92kRp14bIPvu45tVNCj17ArMo23rcYSeQAl8gXml3NZbST6ZreowdEKcbhnCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1ad9f9d49f7eefecf92cf8487af979437be51b33fd39e77f1fa2a467a6d053e4","last_reissued_at":"2026-07-05T10:40:22.669035Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:22.669035Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"R-PRM: Reasoning-Driven Process Reward Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiajun Chen, Junxiao Liu, Shuaijie She, Shujian Huang, Xin Huang, Yifeng Liu","submitted_at":"2025-03-27T09:23:08Z","abstract_excerpt":"Large language models (LLMs) inevitably make mistakes when performing step-by-step mathematical reasoning. Process Reward Models (PRMs) have emerged as a promising solution by evaluating each reasoning step. However, existing PRMs typically output evaluation scores directly, limiting both learning efficiency and evaluation accuracy, which is further exacerbated by the scarcity of annotated data. To address these issues, we propose Reasoning-Driven Process Reward Modeling (R-PRM). First, we leverage stronger LLMs to generate seed data from limited annotations, effectively bootstrapping our mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.21295","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.21295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.21295","created_at":"2026-07-05T10:40:22.669092+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.21295v1","created_at":"2026-07-05T10:40:22.669092+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.21295","created_at":"2026-07-05T10:40:22.669092+00:00"},{"alias_kind":"pith_short_12","alias_value":"DLM7TVE7P3X6","created_at":"2026-07-05T10:40:22.669092+00:00"},{"alias_kind":"pith_short_16","alias_value":"DLM7TVE7P3X6Z6JM","created_at":"2026-07-05T10:40:22.669092+00:00"},{"alias_kind":"pith_short_8","alias_value":"DLM7TVE7","created_at":"2026-07-05T10:40:22.669092+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27369","citing_title":"Reinforcement Learning without Ground-Truth Solutions can Improve LLMs","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12578","citing_title":"MARD: Mirror-Augmented Reasoning Distillation for Mechanism-Level Drug-Drug Interaction Prediction","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15951","citing_title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14703","citing_title":"ToolPRM: Fine-Grained Inference Scaling of Structured Outputs for Function Calling","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15725","citing_title":"Reasoning-targeted Jailbreak Attacks on Large Reasoning Models via Semantic Triggers and Psychological Framing","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19656","citing_title":"Pause or Fabricate? Training Language Models for Grounded Reasoning","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN","json":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN.json","graph_json":"https://pith.science/api/pith-number/DLM7TVE7P3X6Z6JM7BEHV6LZIN/graph.json","events_json":"https://pith.science/api/pith-number/DLM7TVE7P3X6Z6JM7BEHV6LZIN/events.json","paper":"https://pith.science/paper/DLM7TVE7"},"agent_actions":{"view_html":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN","download_json":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN.json","view_paper":"https://pith.science/paper/DLM7TVE7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.21295&json=true","fetch_graph":"https://pith.science/api/pith-number/DLM7TVE7P3X6Z6JM7BEHV6LZIN/graph.json","fetch_events":"https://pith.science/api/pith-number/DLM7TVE7P3X6Z6JM7BEHV6LZIN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN/action/storage_attestation","attest_author":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN/action/author_attestation","sign_citation":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN/action/citation_signature","submit_replication":"https://pith.science/pith/DLM7TVE7P3X6Z6JM7BEHV6LZIN/action/replication_record"}},"created_at":"2026-07-05T10:40:22.669092+00:00","updated_at":"2026-07-05T10:40:22.669092+00:00"}