{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DOGF4GT27OAHXPRDY27VJTOQON","short_pith_number":"pith:DOGF4GT2","schema_version":"1.0","canonical_sha256":"1b8c5e1a7afb807bbe23c6bf54cdd0735f319fe061b98480804afe41333039c4","source":{"kind":"arxiv","id":"2410.05605","version":2},"attestation_state":"computed","paper":{"title":"CodeDPO: Aligning Code Models with Self Generated and Verified Source Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ge Li, Jingjing Xu, Jing Su, Jun Zhang, Kechi Zhang, Yihong Dong, Yongfei Liu, Zhi Jin","submitted_at":"2024-10-08T01:36:15Z","abstract_excerpt":"Code generation models have shown significant potential for programming tasks. However, existing training methods like supervised fine-tuning face key limitations: they do not effectively teach models to prioritize correct over incorrect solutions in ambiguous situations, nor do they effectively optimize the runtime efficiency of the generated code. To address these challenges, we propose CodeDPO, a framework that integrates preference learning into code generation to improve two key code preference factors: code correctness and efficiency. CodeDPO employs a novel dataset construction method, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05605","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-10-08T01:36:15Z","cross_cats_sorted":[],"title_canon_sha256":"6ec50b3c025b397ca65c495bd9b89c9d3dc4ef16ffee82d579d45fea78b86ada","abstract_canon_sha256":"68af8dda1f45009b08d9db412d13ff4f4e795465254fde8715bb9a2daff223e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:49.613969Z","signature_b64":"b1Yw5NZVFDryqpYskQbypjDJ6RAc1CceagxAavZKABMuJxTn+J1/Y5F/s3UqA3lDpORmWtzjqlXZnYYTs9QyDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b8c5e1a7afb807bbe23c6bf54cdd0735f319fe061b98480804afe41333039c4","last_reissued_at":"2026-07-05T11:14:49.613535Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:49.613535Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeDPO: Aligning Code Models with Self Generated and Verified Source Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ge Li, Jingjing Xu, Jing Su, Jun Zhang, Kechi Zhang, Yihong Dong, Yongfei Liu, Zhi Jin","submitted_at":"2024-10-08T01:36:15Z","abstract_excerpt":"Code generation models have shown significant potential for programming tasks. However, existing training methods like supervised fine-tuning face key limitations: they do not effectively teach models to prioritize correct over incorrect solutions in ambiguous situations, nor do they effectively optimize the runtime efficiency of the generated code. To address these challenges, we propose CodeDPO, a framework that integrates preference learning into code generation to improve two key code preference factors: code correctness and efficiency. CodeDPO employs a novel dataset construction method, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05605","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05605/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05605","created_at":"2026-07-05T11:14:49.613591+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05605v2","created_at":"2026-07-05T11:14:49.613591+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05605","created_at":"2026-07-05T11:14:49.613591+00:00"},{"alias_kind":"pith_short_12","alias_value":"DOGF4GT27OAH","created_at":"2026-07-05T11:14:49.613591+00:00"},{"alias_kind":"pith_short_16","alias_value":"DOGF4GT27OAHXPRD","created_at":"2026-07-05T11:14:49.613591+00:00"},{"alias_kind":"pith_short_8","alias_value":"DOGF4GT2","created_at":"2026-07-05T11:14:49.613591+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06826","citing_title":"SkelDPO: A Skeleton-Guided Direct Preference Optimization Framework for Efficient Code Generation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06821","citing_title":"Chiseling Out Efficiency: Structured Skeleton Supervision for Efficient Code Generation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03489","citing_title":"Learn from Your Mistakes: Tree-like Self-Play for Secure Code LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07605","citing_title":"Fine-R1: Make Multi-modal LLMs Excel in Fine-Grained Visual Recognition by Chain-of-Thought Reasoning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2503.01785","citing_title":"Visual-RFT: Visual Reinforcement Fine-Tuning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11974","citing_title":"Towards Order Fairness: Mitigating LLMs Order Sensitivity through Dual Group Advantage Optimization","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05560","citing_title":"An Iterative Test-and-Repair Framework for Competitive Code Generation","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON","json":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON.json","graph_json":"https://pith.science/api/pith-number/DOGF4GT27OAHXPRDY27VJTOQON/graph.json","events_json":"https://pith.science/api/pith-number/DOGF4GT27OAHXPRDY27VJTOQON/events.json","paper":"https://pith.science/paper/DOGF4GT2"},"agent_actions":{"view_html":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON","download_json":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON.json","view_paper":"https://pith.science/paper/DOGF4GT2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05605&json=true","fetch_graph":"https://pith.science/api/pith-number/DOGF4GT27OAHXPRDY27VJTOQON/graph.json","fetch_events":"https://pith.science/api/pith-number/DOGF4GT27OAHXPRDY27VJTOQON/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON/action/storage_attestation","attest_author":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON/action/author_attestation","sign_citation":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON/action/citation_signature","submit_replication":"https://pith.science/pith/DOGF4GT27OAHXPRDY27VJTOQON/action/replication_record"}},"created_at":"2026-07-05T11:14:49.613591+00:00","updated_at":"2026-07-05T11:14:49.613591+00:00"}