{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:55TREVZFMXAWHA5K336VN5IK6B","short_pith_number":"pith:55TREVZF","schema_version":"1.0","canonical_sha256":"ef6712572565c16383aadefd56f50af04a9af3d36a7d38a533ea259cdf1f1617","source":{"kind":"arxiv","id":"2412.15118","version":2},"attestation_state":"computed","paper":{"title":"Reasoning Through Execution: Unifying Process and Outcome Rewards for Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SE"],"primary_cat":"cs.CL","authors_text":"Jindong Wang, Shikun Zhang, Wei Ye, Weizheng Gu, Xingru Jiang, Yidong Wang, Zhengran Zeng, Zhuohao Yu","submitted_at":"2024-12-19T17:59:42Z","abstract_excerpt":"Large Language Models excel at code generation yet struggle with complex programming tasks that demand sophisticated reasoning. To bridge this gap, traditional process supervision relies on learned reward models requiring costly training data and suffering from reward misalignment, while outcome supervision fails for complex tasks needing coordinated intermediate steps. We introduce Outcome Refining Process Supervision, which unifies process and outcome supervision by leveraging executable verification: a tree-structured search framework generates strategic alternatives, profiles execution met"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15118","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-19T17:59:42Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SE"],"title_canon_sha256":"af17e04092ef225d3217ba00914a6bcadc27f3f7f40ebeeccbe9bd14d9a0083e","abstract_canon_sha256":"b8f07faadba0a118ea07e0bd93ad11332a8d7487e62edae09574a8610f187b60"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:53.501177Z","signature_b64":"dEHFlztkFeR5yeh5wVdzHutIsPIh+60Tee36v3eu53N5yqBAi2tWnzk4wWfEgE7u9ODxLIEglfPy+dQS5pM0Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef6712572565c16383aadefd56f50af04a9af3d36a7d38a533ea259cdf1f1617","last_reissued_at":"2026-07-05T11:16:53.500641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:53.500641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning Through Execution: Unifying Process and Outcome Rewards for Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SE"],"primary_cat":"cs.CL","authors_text":"Jindong Wang, Shikun Zhang, Wei Ye, Weizheng Gu, Xingru Jiang, Yidong Wang, Zhengran Zeng, Zhuohao Yu","submitted_at":"2024-12-19T17:59:42Z","abstract_excerpt":"Large Language Models excel at code generation yet struggle with complex programming tasks that demand sophisticated reasoning. To bridge this gap, traditional process supervision relies on learned reward models requiring costly training data and suffering from reward misalignment, while outcome supervision fails for complex tasks needing coordinated intermediate steps. We introduce Outcome Refining Process Supervision, which unifies process and outcome supervision by leveraging executable verification: a tree-structured search framework generates strategic alternatives, profiles execution met"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15118","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15118/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15118","created_at":"2026-07-05T11:16:53.500712+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15118v2","created_at":"2026-07-05T11:16:53.500712+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15118","created_at":"2026-07-05T11:16:53.500712+00:00"},{"alias_kind":"pith_short_12","alias_value":"55TREVZFMXAW","created_at":"2026-07-05T11:16:53.500712+00:00"},{"alias_kind":"pith_short_16","alias_value":"55TREVZFMXAWHA5K","created_at":"2026-07-05T11:16:53.500712+00:00"},{"alias_kind":"pith_short_8","alias_value":"55TREVZF","created_at":"2026-07-05T11:16:53.500712+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18747","citing_title":"Code as Agent Harness","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08468","citing_title":"PYTHALAB-MERA: Validation-Grounded Memory, Retrieval, and Acceptance Control for Frozen-LLM Coding Agents","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07248","citing_title":"PaT: Planning-after-Trial for Efficient Test-Time Code Generation","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B","json":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B.json","graph_json":"https://pith.science/api/pith-number/55TREVZFMXAWHA5K336VN5IK6B/graph.json","events_json":"https://pith.science/api/pith-number/55TREVZFMXAWHA5K336VN5IK6B/events.json","paper":"https://pith.science/paper/55TREVZF"},"agent_actions":{"view_html":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B","download_json":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B.json","view_paper":"https://pith.science/paper/55TREVZF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15118&json=true","fetch_graph":"https://pith.science/api/pith-number/55TREVZFMXAWHA5K336VN5IK6B/graph.json","fetch_events":"https://pith.science/api/pith-number/55TREVZFMXAWHA5K336VN5IK6B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B/action/storage_attestation","attest_author":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B/action/author_attestation","sign_citation":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B/action/citation_signature","submit_replication":"https://pith.science/pith/55TREVZFMXAWHA5K336VN5IK6B/action/replication_record"}},"created_at":"2026-07-05T11:16:53.500712+00:00","updated_at":"2026-07-05T11:16:53.500712+00:00"}