{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NGVGQZ7ERNIVRBMPPXRGOHGVJO","short_pith_number":"pith:NGVGQZ7E","schema_version":"1.0","canonical_sha256":"69aa6867e48b5158858f7de2671cd54b98e24766c65c822609f8af5248583f23","source":{"kind":"arxiv","id":"2406.05892","version":1},"attestation_state":"computed","paper":{"title":"Security Vulnerability Detection with Multitask Self-Instructed Fine-Tuning of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.SE"],"primary_cat":"cs.CR","authors_text":"Aidan Z.H. Yang, Claire Le Goues, Haoye Tian, He Ye, Ruben Martins","submitted_at":"2024-06-09T19:18:05Z","abstract_excerpt":"Software security vulnerabilities allow attackers to perform malicious activities to disrupt software operations. Recent Transformer-based language models have significantly advanced vulnerability detection, surpassing the capabilities of static analysis based deep learning models. However, language models trained solely on code tokens do not capture either the explanation of vulnerability type or the data flow structure information of code, both of which are crucial for vulnerability detection. We propose a novel technique that integrates a multitask sequence-to-sequence LLM with pro-gram con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05892","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-06-09T19:18:05Z","cross_cats_sorted":["cs.LG","cs.SE"],"title_canon_sha256":"9d3092a1986497041f96a87377e7b7647d4bab6884d21d6f1ef44992131901ff","abstract_canon_sha256":"7155d7bbcabf68962e704e9791add5a4a3617e7d975e7ef67cdc551513ac6f5b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:34.991023Z","signature_b64":"v424HqFU9gd0KmZlq0dKW1j7L0Zl5eS6mR1p3iICoRuq5GyLabfy6GwaE9ajXcmzhkDXovLIrYwX78GDcRgjDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69aa6867e48b5158858f7de2671cd54b98e24766c65c822609f8af5248583f23","last_reissued_at":"2026-07-05T08:29:34.990611Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:34.990611Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Security Vulnerability Detection with Multitask Self-Instructed Fine-Tuning of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.SE"],"primary_cat":"cs.CR","authors_text":"Aidan Z.H. Yang, Claire Le Goues, Haoye Tian, He Ye, Ruben Martins","submitted_at":"2024-06-09T19:18:05Z","abstract_excerpt":"Software security vulnerabilities allow attackers to perform malicious activities to disrupt software operations. Recent Transformer-based language models have significantly advanced vulnerability detection, surpassing the capabilities of static analysis based deep learning models. However, language models trained solely on code tokens do not capture either the explanation of vulnerability type or the data flow structure information of code, both of which are crucial for vulnerability detection. We propose a novel technique that integrates a multitask sequence-to-sequence LLM with pro-gram con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05892","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05892/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05892","created_at":"2026-07-05T08:29:34.990665+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05892v1","created_at":"2026-07-05T08:29:34.990665+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05892","created_at":"2026-07-05T08:29:34.990665+00:00"},{"alias_kind":"pith_short_12","alias_value":"NGVGQZ7ERNIV","created_at":"2026-07-05T08:29:34.990665+00:00"},{"alias_kind":"pith_short_16","alias_value":"NGVGQZ7ERNIVRBMP","created_at":"2026-07-05T08:29:34.990665+00:00"},{"alias_kind":"pith_short_8","alias_value":"NGVGQZ7E","created_at":"2026-07-05T08:29:34.990665+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11015","citing_title":"DCVD: Dual-Channel Cross-Modal Fusion for Joint Vulnerability Detection and Localization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25599","citing_title":"PLMGH: What Matters in PLM-GNN Hybrids for Code Classification and Vulnerability Detection","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10767","citing_title":"VulWeaver: Weaving Broken Semantics for Grounded Vulnerability Detection","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19031","citing_title":"SAGE: Signal-Amplified Guided Embeddings for LLM-based Vulnerability Detection","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO","json":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO.json","graph_json":"https://pith.science/api/pith-number/NGVGQZ7ERNIVRBMPPXRGOHGVJO/graph.json","events_json":"https://pith.science/api/pith-number/NGVGQZ7ERNIVRBMPPXRGOHGVJO/events.json","paper":"https://pith.science/paper/NGVGQZ7E"},"agent_actions":{"view_html":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO","download_json":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO.json","view_paper":"https://pith.science/paper/NGVGQZ7E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05892&json=true","fetch_graph":"https://pith.science/api/pith-number/NGVGQZ7ERNIVRBMPPXRGOHGVJO/graph.json","fetch_events":"https://pith.science/api/pith-number/NGVGQZ7ERNIVRBMPPXRGOHGVJO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO/action/storage_attestation","attest_author":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO/action/author_attestation","sign_citation":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO/action/citation_signature","submit_replication":"https://pith.science/pith/NGVGQZ7ERNIVRBMPPXRGOHGVJO/action/replication_record"}},"created_at":"2026-07-05T08:29:34.990665+00:00","updated_at":"2026-07-05T08:29:34.990665+00:00"}