{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:NVAPJTJZ66BKQCEL3JXRYIUQJI","short_pith_number":"pith:NVAPJTJZ","canonical_record":{"source":{"id":"2607.26457","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T04:16:33Z","cross_cats_sorted":[],"title_canon_sha256":"39322927061a3b9e1d48ad97324ecb6760554239603f6eba176dbb6a3370d528","abstract_canon_sha256":"08e7f4754355f5740521028278f4b019bce0b9564ae2e5178cfaf9bd83fe6d1d"},"schema_version":"1.0"},"canonical_sha256":"6d40f4cd39f782a8088bda6f1c22904a04d9973fa7144296798371b6b3206eae","source":{"kind":"arxiv","id":"2607.26457","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.26457","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"arxiv_version","alias_value":"2607.26457v1","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.26457","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"pith_short_12","alias_value":"NVAPJTJZ66BK","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"pith_short_16","alias_value":"NVAPJTJZ66BKQCEL","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"pith_short_8","alias_value":"NVAPJTJZ","created_at":"2026-07-30T01:20:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:NVAPJTJZ66BKQCEL3JXRYIUQJI","target":"record","payload":{"canonical_record":{"source":{"id":"2607.26457","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T04:16:33Z","cross_cats_sorted":[],"title_canon_sha256":"39322927061a3b9e1d48ad97324ecb6760554239603f6eba176dbb6a3370d528","abstract_canon_sha256":"08e7f4754355f5740521028278f4b019bce0b9564ae2e5178cfaf9bd83fe6d1d"},"schema_version":"1.0"},"canonical_sha256":"6d40f4cd39f782a8088bda6f1c22904a04d9973fa7144296798371b6b3206eae","receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d40f4cd39f782a8088bda6f1c22904a04d9973fa7144296798371b6b3206eae","last_reissued_at":"2026-07-30T01:20:33.491853Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-30T01:20:33.491853Z"},"source_kind":"arxiv","source_id":"2607.26457","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-30T01:20:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KkYD3tcQwlK4k5Xn02Suu5saB3PwykFWnkdCCypS7G7uEYiGZIZYu3a1RlHpiKIcHQ39XwLiICk+XbUIKVXUBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T04:07:14.925181Z"},"content_sha256":"15adadb871ce3f341e2f505b3f9421d8e20a2a2064312a78bdbee67babd8ce7a","schema_version":"1.0","event_id":"sha256:15adadb871ce3f341e2f505b3f9421d8e20a2a2064312a78bdbee67babd8ce7a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:NVAPJTJZ66BKQCEL3JXRYIUQJI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"DHRCL:Training Code LLMs with Dense Hierarchical Rewards and Curriculum Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Hui Cheng, Shuhang Wang, Ziming Li","submitted_at":"2026-07-29T04:16:33Z","abstract_excerpt":"Reinforcement learning is a natural post-training paradigm for code-oriented large language models because generated programs can be evaluated through parsing, execution, unit tests, and structural analysis.However, existing methods often rely on sparse outcome rewards or statically combine heterogeneous dense signals, even though syntax validity, executability, functional correctness, and structural organization describe different and progressively dependent programming capabilities. We propose DHRCL, a reinforcement learning framework with Dense Hierarchical Rewards and Curriculum Learning. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.26457","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.26457/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-30T01:20:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VzO3UsL8XA5YQu4wYollmC0yR6UQCDJGdX5W4wXQmYi3ZFluuBTtQPMl+2P6UbylH7FaE7EVtFEF4JEhk2fvDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T04:07:14.925673Z"},"content_sha256":"9d90897f7e764207039cb673051c3f8ba80744321909dbf1c8e019595074b8e0","schema_version":"1.0","event_id":"sha256:9d90897f7e764207039cb673051c3f8ba80744321909dbf1c8e019595074b8e0"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NVAPJTJZ66BKQCEL3JXRYIUQJI/bundle.json","state_url":"https://pith.science/pith/NVAPJTJZ66BKQCEL3JXRYIUQJI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NVAPJTJZ66BKQCEL3JXRYIUQJI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T04:07:14Z","links":{"resolver":"https://pith.science/pith/NVAPJTJZ66BKQCEL3JXRYIUQJI","bundle":"https://pith.science/pith/NVAPJTJZ66BKQCEL3JXRYIUQJI/bundle.json","state":"https://pith.science/pith/NVAPJTJZ66BKQCEL3JXRYIUQJI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NVAPJTJZ66BKQCEL3JXRYIUQJI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:NVAPJTJZ66BKQCEL3JXRYIUQJI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"08e7f4754355f5740521028278f4b019bce0b9564ae2e5178cfaf9bd83fe6d1d","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T04:16:33Z","title_canon_sha256":"39322927061a3b9e1d48ad97324ecb6760554239603f6eba176dbb6a3370d528"},"schema_version":"1.0","source":{"id":"2607.26457","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.26457","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"arxiv_version","alias_value":"2607.26457v1","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.26457","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"pith_short_12","alias_value":"NVAPJTJZ66BK","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"pith_short_16","alias_value":"NVAPJTJZ66BKQCEL","created_at":"2026-07-30T01:20:33Z"},{"alias_kind":"pith_short_8","alias_value":"NVAPJTJZ","created_at":"2026-07-30T01:20:33Z"}],"graph_snapshots":[{"event_id":"sha256:9d90897f7e764207039cb673051c3f8ba80744321909dbf1c8e019595074b8e0","target":"graph","created_at":"2026-07-30T01:20:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.26457/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning is a natural post-training paradigm for code-oriented large language models because generated programs can be evaluated through parsing, execution, unit tests, and structural analysis.However, existing methods often rely on sparse outcome rewards or statically combine heterogeneous dense signals, even though syntax validity, executability, functional correctness, and structural organization describe different and progressively dependent programming capabilities. We propose DHRCL, a reinforcement learning framework with Dense Hierarchical Rewards and Curriculum Learning. ","authors_text":"Hui Cheng, Shuhang Wang, Ziming Li","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T04:16:33Z","title":"DHRCL:Training Code LLMs with Dense Hierarchical Rewards and Curriculum Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.26457","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:15adadb871ce3f341e2f505b3f9421d8e20a2a2064312a78bdbee67babd8ce7a","target":"record","created_at":"2026-07-30T01:20:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"08e7f4754355f5740521028278f4b019bce0b9564ae2e5178cfaf9bd83fe6d1d","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-29T04:16:33Z","title_canon_sha256":"39322927061a3b9e1d48ad97324ecb6760554239603f6eba176dbb6a3370d528"},"schema_version":"1.0","source":{"id":"2607.26457","kind":"arxiv","version":1}},"canonical_sha256":"6d40f4cd39f782a8088bda6f1c22904a04d9973fa7144296798371b6b3206eae","receipt":{"builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6d40f4cd39f782a8088bda6f1c22904a04d9973fa7144296798371b6b3206eae","first_computed_at":"2026-07-30T01:20:33.491853Z","kind":"pith_receipt","last_reissued_at":"2026-07-30T01:20:33.491853Z","receipt_version":"0.3","signature_status":"unsigned_v0"},"source_id":"2607.26457","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:15adadb871ce3f341e2f505b3f9421d8e20a2a2064312a78bdbee67babd8ce7a","sha256:9d90897f7e764207039cb673051c3f8ba80744321909dbf1c8e019595074b8e0"],"state_sha256":"0d36c32fb8de6bccc79ec5afe68fd9d1094ffaf2ec1256b3c4d06fc7a7e29dd9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"77Qimifzqi3L2jQyELNcpZjjYlbG6cRB5ydPzPMDiHtAVtuPgQ1NsJL+Soy7VERI+hKkRykm6coJ/5W74SSmAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T04:07:14.929233Z","bundle_sha256":"3c8f0e4436953915e399310cf31cbe60e0b3fbf2b1e36ab75846dd31d82918c1"}}