{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GNCYCXSPR5BVX4YYSMKLX3MSX7","short_pith_number":"pith:GNCYCXSP","schema_version":"1.0","canonical_sha256":"3345815e4f8f435bf3189314bbed92bfc903732e4b5d7f88d0d4f5bfe385dda1","source":{"kind":"arxiv","id":"2403.05286","version":3},"attestation_state":"computed","paper":{"title":"LLM4Decompile: Decompiling Binary Code with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.PL","authors_text":"Hanzhuo Tan, Jing Li, Qi Luo, Yuqun Zhang","submitted_at":"2024-03-08T13:10:59Z","abstract_excerpt":"Decompilation aims to convert binary code to high-level source code, but traditional tools like Ghidra often produce results that are difficult to read and execute. Motivated by the advancements in Large Language Models (LLMs), we propose LLM4Decompile, the first and largest open-source LLM series (1.3B to 33B) trained to decompile binary code. We optimize the LLM training process and introduce the LLM4Decompile-End models to decompile binary directly. The resulting models significantly outperform GPT-4o and Ghidra on the HumanEval and ExeBench benchmarks by over 100% in terms of re-executabil"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.05286","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.PL","submitted_at":"2024-03-08T13:10:59Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"cac6218d1de5e7411d08bdf91fd3fed774156d62c87bdef2119feaf017b98103","abstract_canon_sha256":"5509fabc13eb47edecaf7b256fe367ed9f51f7d90c3df8602c7aa51203333063"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:19.966845Z","signature_b64":"VmI/COSwY4fB0uZjXCbGEeAr2aTxtyr+NJlWeJHFS0PNVXMvFteCZqTsxezOw83xhAp8yxpzu2fDwm/GYFXHAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3345815e4f8f435bf3189314bbed92bfc903732e4b5d7f88d0d4f5bfe385dda1","last_reissued_at":"2026-07-05T11:48:19.966340Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:19.966340Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM4Decompile: Decompiling Binary Code with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.PL","authors_text":"Hanzhuo Tan, Jing Li, Qi Luo, Yuqun Zhang","submitted_at":"2024-03-08T13:10:59Z","abstract_excerpt":"Decompilation aims to convert binary code to high-level source code, but traditional tools like Ghidra often produce results that are difficult to read and execute. Motivated by the advancements in Large Language Models (LLMs), we propose LLM4Decompile, the first and largest open-source LLM series (1.3B to 33B) trained to decompile binary code. We optimize the LLM training process and introduce the LLM4Decompile-End models to decompile binary directly. The resulting models significantly outperform GPT-4o and Ghidra on the HumanEval and ExeBench benchmarks by over 100% in terms of re-executabil"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.05286","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.05286/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.05286","created_at":"2026-07-05T11:48:19.966398+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.05286v3","created_at":"2026-07-05T11:48:19.966398+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.05286","created_at":"2026-07-05T11:48:19.966398+00:00"},{"alias_kind":"pith_short_12","alias_value":"GNCYCXSPR5BV","created_at":"2026-07-05T11:48:19.966398+00:00"},{"alias_kind":"pith_short_16","alias_value":"GNCYCXSPR5BVX4YY","created_at":"2026-07-05T11:48:19.966398+00:00"},{"alias_kind":"pith_short_8","alias_value":"GNCYCXSP","created_at":"2026-07-05T11:48:19.966398+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06838","citing_title":"LLM Agent-Assisted Reverse Engineering with Quantitative Readability Metrics","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00635","citing_title":"Checked Program Recovery from Execution Video: A Sound Oracle for Untrusted Generators","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29155","citing_title":"OASIF: An Efficient Obfuscation-Aware Self-Improving Framework for LLM-Based Assembly Code Instruction Following and Comprehension","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07243","citing_title":"Beyond the Edge of Function: Unraveling the Patterns of Type Recovery in Binary Code","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01763","citing_title":"Context-Guided Decompilation: A Step Towards Re-executability","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11501","citing_title":"Decaf: Improving Neural Decompilation with Automatic Feedback and Search","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23940","citing_title":"Constraint-Guided Multi-Agent Decompilation for Executable Binary Recovery","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05000","citing_title":"Agentic Vulnerability Reasoning on COTS Binaries","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08083","citing_title":"Can LLMs Deobfuscate Binary Code? A Systematic Analysis of Large Language Models into Pseudocode Deobfuscation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12913","citing_title":"CoDe-R: Refining Decompiler Output with LLMs via Rationale Guidance and Adaptive Inference","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7","json":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7.json","graph_json":"https://pith.science/api/pith-number/GNCYCXSPR5BVX4YYSMKLX3MSX7/graph.json","events_json":"https://pith.science/api/pith-number/GNCYCXSPR5BVX4YYSMKLX3MSX7/events.json","paper":"https://pith.science/paper/GNCYCXSP"},"agent_actions":{"view_html":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7","download_json":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7.json","view_paper":"https://pith.science/paper/GNCYCXSP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.05286&json=true","fetch_graph":"https://pith.science/api/pith-number/GNCYCXSPR5BVX4YYSMKLX3MSX7/graph.json","fetch_events":"https://pith.science/api/pith-number/GNCYCXSPR5BVX4YYSMKLX3MSX7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7/action/storage_attestation","attest_author":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7/action/author_attestation","sign_citation":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7/action/citation_signature","submit_replication":"https://pith.science/pith/GNCYCXSPR5BVX4YYSMKLX3MSX7/action/replication_record"}},"created_at":"2026-07-05T11:48:19.966398+00:00","updated_at":"2026-07-05T11:48:19.966398+00:00"}