{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SB3HQIUMWF3DXYYETCC6DJCEGJ","short_pith_number":"pith:SB3HQIUM","schema_version":"1.0","canonical_sha256":"907678228cb1763be3049885e1a444327718160f923d90846d19a0e594766ad6","source":{"kind":"arxiv","id":"2502.01715","version":1},"attestation_state":"computed","paper":{"title":"Process-Supervised Reinforcement Learning for Code Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Hua Huang, Ting Zhang, Wenbin Jiang, Yufan Ye","submitted_at":"2025-02-03T16:22:06Z","abstract_excerpt":"Existing reinforcement learning strategies based on outcome supervision have proven effective in enhancing the performance of large language models(LLMs) for code generation. While reinforcement learning based on process supervision has shown great promise in handling multi-step reasoning tasks, its effectiveness in code generation remains largely underexplored and underjustified. The primary obstacle stems from the resource-intensive nature of constructing high-quality process-supervised data, which demands substantial human expertise and computational resources. In response to this challenge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01715","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-02-03T16:22:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b4bcd547b10a300a848d2791a5703ca9269af9065803c8660ddf673b5da304f1","abstract_canon_sha256":"96d97b176a2484445868f5da6895e5be11df915e971a0331bca07d87c6ac2020"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:18.166787Z","signature_b64":"7B+bu7D/2s0i21jYwBkqlygDQrqsiBepjAiTHOOaPcThm16PfgXx2umIWN9EQvPrZ35QoMjNemXr9zJJUh08Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"907678228cb1763be3049885e1a444327718160f923d90846d19a0e594766ad6","last_reissued_at":"2026-07-05T10:09:18.166336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:18.166336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Process-Supervised Reinforcement Learning for Code Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Hua Huang, Ting Zhang, Wenbin Jiang, Yufan Ye","submitted_at":"2025-02-03T16:22:06Z","abstract_excerpt":"Existing reinforcement learning strategies based on outcome supervision have proven effective in enhancing the performance of large language models(LLMs) for code generation. While reinforcement learning based on process supervision has shown great promise in handling multi-step reasoning tasks, its effectiveness in code generation remains largely underexplored and underjustified. The primary obstacle stems from the resource-intensive nature of constructing high-quality process-supervised data, which demands substantial human expertise and computational resources. In response to this challenge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01715","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01715/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01715","created_at":"2026-07-05T10:09:18.166399+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01715v1","created_at":"2026-07-05T10:09:18.166399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01715","created_at":"2026-07-05T10:09:18.166399+00:00"},{"alias_kind":"pith_short_12","alias_value":"SB3HQIUMWF3D","created_at":"2026-07-05T10:09:18.166399+00:00"},{"alias_kind":"pith_short_16","alias_value":"SB3HQIUMWF3DXYYE","created_at":"2026-07-05T10:09:18.166399+00:00"},{"alias_kind":"pith_short_8","alias_value":"SB3HQIUM","created_at":"2026-07-05T10:09:18.166399+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28707","citing_title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29697","citing_title":"Beyond Trajectory Rewards: Step-level Credit Assignment for Agentic Search via Graph Modeling","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01799","citing_title":"TestDecision: Sequential Test Suite Generation via Greedy Optimization and Reinforcement Learning","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04066","citing_title":"Adapt to Thrive! Adaptive Power-Mean Policy Optimization for Improved LLM Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ","json":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ.json","graph_json":"https://pith.science/api/pith-number/SB3HQIUMWF3DXYYETCC6DJCEGJ/graph.json","events_json":"https://pith.science/api/pith-number/SB3HQIUMWF3DXYYETCC6DJCEGJ/events.json","paper":"https://pith.science/paper/SB3HQIUM"},"agent_actions":{"view_html":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ","download_json":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ.json","view_paper":"https://pith.science/paper/SB3HQIUM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01715&json=true","fetch_graph":"https://pith.science/api/pith-number/SB3HQIUMWF3DXYYETCC6DJCEGJ/graph.json","fetch_events":"https://pith.science/api/pith-number/SB3HQIUMWF3DXYYETCC6DJCEGJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ/action/storage_attestation","attest_author":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ/action/author_attestation","sign_citation":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ/action/citation_signature","submit_replication":"https://pith.science/pith/SB3HQIUMWF3DXYYETCC6DJCEGJ/action/replication_record"}},"created_at":"2026-07-05T10:09:18.166399+00:00","updated_at":"2026-07-05T10:09:18.166399+00:00"}