{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4NHFGSD5U4NIZLZTX2G5ESGYU3","short_pith_number":"pith:4NHFGSD5","schema_version":"1.0","canonical_sha256":"e34e53487da71a8caf33be8dd248d8a6df9ab80d212bd6801cb22f7aec8883f9","source":{"kind":"arxiv","id":"2409.06446","version":1},"attestation_state":"computed","paper":{"title":"HexaCoder: Secure Code Generation via Oracle-Guided Synthetic Training Data","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.SE"],"primary_cat":"cs.CR","authors_text":"Hossein Hajipour, Lea Sch\\\"onherr, Mario Fritz, Thorsten Holz","submitted_at":"2024-09-10T12:01:43Z","abstract_excerpt":"Large language models (LLMs) have shown great potential for automatic code generation and form the basis for various tools such as GitHub Copilot. However, recent studies highlight that many LLM-generated code contains serious security vulnerabilities. While previous work tries to address this by training models that generate secure code, these attempts remain constrained by limited access to training data and labor-intensive data preparation.\n  In this paper, we introduce HexaCoder, a novel approach to enhance the ability of LLMs to generate secure codes by automatically synthesizing secure c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.06446","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CR","submitted_at":"2024-09-10T12:01:43Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","cs.SE"],"title_canon_sha256":"213b96fc9db1f0875513444309415b0d34c2802eca75dc41f98fa10d08ab22c0","abstract_canon_sha256":"e180fffe79cd2ef51c2cf794d5467e2bd8687aa9d39ea916ff26748ca5692029"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:05:24.564979Z","signature_b64":"FMWx6Il3UabMzighlr/GmWlGzBGWvOudwz/018ZKaBBH8hNumga0r2fDxTlXG0tJz12iP0NObdlm5y4p2g1WAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e34e53487da71a8caf33be8dd248d8a6df9ab80d212bd6801cb22f7aec8883f9","last_reissued_at":"2026-07-05T09:05:24.564491Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:05:24.564491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HexaCoder: Secure Code Generation via Oracle-Guided Synthetic Training Data","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.SE"],"primary_cat":"cs.CR","authors_text":"Hossein Hajipour, Lea Sch\\\"onherr, Mario Fritz, Thorsten Holz","submitted_at":"2024-09-10T12:01:43Z","abstract_excerpt":"Large language models (LLMs) have shown great potential for automatic code generation and form the basis for various tools such as GitHub Copilot. However, recent studies highlight that many LLM-generated code contains serious security vulnerabilities. While previous work tries to address this by training models that generate secure code, these attempts remain constrained by limited access to training data and labor-intensive data preparation.\n  In this paper, we introduce HexaCoder, a novel approach to enhance the ability of LLMs to generate secure codes by automatically synthesizing secure c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06446","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.06446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.06446","created_at":"2026-07-05T09:05:24.564549+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.06446v1","created_at":"2026-07-05T09:05:24.564549+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06446","created_at":"2026-07-05T09:05:24.564549+00:00"},{"alias_kind":"pith_short_12","alias_value":"4NHFGSD5U4NI","created_at":"2026-07-05T09:05:24.564549+00:00"},{"alias_kind":"pith_short_16","alias_value":"4NHFGSD5U4NIZLZT","created_at":"2026-07-05T09:05:24.564549+00:00"},{"alias_kind":"pith_short_8","alias_value":"4NHFGSD5","created_at":"2026-07-05T09:05:24.564549+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25195","citing_title":"SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03489","citing_title":"Learn from Your Mistakes: Tree-like Self-Play for Secure Code LLMs","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3","json":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3.json","graph_json":"https://pith.science/api/pith-number/4NHFGSD5U4NIZLZTX2G5ESGYU3/graph.json","events_json":"https://pith.science/api/pith-number/4NHFGSD5U4NIZLZTX2G5ESGYU3/events.json","paper":"https://pith.science/paper/4NHFGSD5"},"agent_actions":{"view_html":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3","download_json":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3.json","view_paper":"https://pith.science/paper/4NHFGSD5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.06446&json=true","fetch_graph":"https://pith.science/api/pith-number/4NHFGSD5U4NIZLZTX2G5ESGYU3/graph.json","fetch_events":"https://pith.science/api/pith-number/4NHFGSD5U4NIZLZTX2G5ESGYU3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3/action/storage_attestation","attest_author":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3/action/author_attestation","sign_citation":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3/action/citation_signature","submit_replication":"https://pith.science/pith/4NHFGSD5U4NIZLZTX2G5ESGYU3/action/replication_record"}},"created_at":"2026-07-05T09:05:24.564549+00:00","updated_at":"2026-07-05T09:05:24.564549+00:00"}