{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:5GAD77AGJKBBZFYM4E5RZRF73X","short_pith_number":"pith:5GAD77AG","schema_version":"1.0","canonical_sha256":"e9803ffc064a821c970ce13b1cc4bfddcef31e7df161e8dcd63b1bd5c53c2f4b","source":{"kind":"arxiv","id":"2607.05199","version":1},"attestation_state":"computed","paper":{"title":"Reason, Reward, Refine: Step-Level Errors Corrections with Structured Feedback for Physics Reasoning in Small Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dhruv Jain, Rajiv Ratn Shah, Raj Jaiswal, Rishabh Dhawan, Shin'ichi Satoh, Sree Krishna Uppalapati, Tanuja Ganu","submitted_at":"2026-07-06T15:16:10Z","abstract_excerpt":"Physics reasoning fails structurally in small language models: an error at any step propagates forward, corrupting every inference that follows. Limited domain knowledge, hallucination under multi-step derivation, and distributional sensitivity compound this failure. We propose a step-level reward framework that identifies the first reasoning error, generates targeted structured feedback, and trains the model to revise its solution via policy gradient with KL regularization, without exposing it to ground truth solutions as generation targets. Unlike annotation-dependent step-level methods, no "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.05199","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-06T15:16:10Z","cross_cats_sorted":[],"title_canon_sha256":"7a9dcf2734bb91af4fc675ecc3ca6415e30874ad8a25ee070b2f33ec3f9ee1fb","abstract_canon_sha256":"cb73d3565515853e88bbff36edc7b53d0be99af11aba05dd29633bf6fc599e46"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T03:19:19.543027Z","signature_b64":"Cw40564PIA1amqDpMK1p79nqsTrFnXdckIEbD88IlafJPVaETwaueWUl02gQJ++xeR34B8+KsSEnXKVpWkgeAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e9803ffc064a821c970ce13b1cc4bfddcef31e7df161e8dcd63b1bd5c53c2f4b","last_reissued_at":"2026-07-07T03:19:19.542538Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T03:19:19.542538Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reason, Reward, Refine: Step-Level Errors Corrections with Structured Feedback for Physics Reasoning in Small Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dhruv Jain, Rajiv Ratn Shah, Raj Jaiswal, Rishabh Dhawan, Shin'ichi Satoh, Sree Krishna Uppalapati, Tanuja Ganu","submitted_at":"2026-07-06T15:16:10Z","abstract_excerpt":"Physics reasoning fails structurally in small language models: an error at any step propagates forward, corrupting every inference that follows. Limited domain knowledge, hallucination under multi-step derivation, and distributional sensitivity compound this failure. We propose a step-level reward framework that identifies the first reasoning error, generates targeted structured feedback, and trains the model to revise its solution via policy gradient with KL regularization, without exposing it to ground truth solutions as generation targets. Unlike annotation-dependent step-level methods, no "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.05199","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.05199/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.05199","created_at":"2026-07-07T03:19:19.542595+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.05199v1","created_at":"2026-07-07T03:19:19.542595+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.05199","created_at":"2026-07-07T03:19:19.542595+00:00"},{"alias_kind":"pith_short_12","alias_value":"5GAD77AGJKBB","created_at":"2026-07-07T03:19:19.542595+00:00"},{"alias_kind":"pith_short_16","alias_value":"5GAD77AGJKBBZFYM","created_at":"2026-07-07T03:19:19.542595+00:00"},{"alias_kind":"pith_short_8","alias_value":"5GAD77AG","created_at":"2026-07-07T03:19:19.542595+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X","json":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X.json","graph_json":"https://pith.science/api/pith-number/5GAD77AGJKBBZFYM4E5RZRF73X/graph.json","events_json":"https://pith.science/api/pith-number/5GAD77AGJKBBZFYM4E5RZRF73X/events.json","paper":"https://pith.science/paper/5GAD77AG"},"agent_actions":{"view_html":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X","download_json":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X.json","view_paper":"https://pith.science/paper/5GAD77AG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.05199&json=true","fetch_graph":"https://pith.science/api/pith-number/5GAD77AGJKBBZFYM4E5RZRF73X/graph.json","fetch_events":"https://pith.science/api/pith-number/5GAD77AGJKBBZFYM4E5RZRF73X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X/action/storage_attestation","attest_author":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X/action/author_attestation","sign_citation":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X/action/citation_signature","submit_replication":"https://pith.science/pith/5GAD77AGJKBBZFYM4E5RZRF73X/action/replication_record"}},"created_at":"2026-07-07T03:19:19.542595+00:00","updated_at":"2026-07-07T03:19:19.542595+00:00"}