{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:75EUI3OGESA7XSJ6ZHFEIQD3P3","short_pith_number":"pith:75EUI3OG","schema_version":"1.0","canonical_sha256":"ff49446dc62481fbc93ec9ca44407b7eff9a1f348a9a1c2bd07631e2bc855331","source":{"kind":"arxiv","id":"2305.14710","version":2},"attestation_state":"computed","paper":{"title":"Instructions as Backdoors: Backdoor Vulnerabilities of Instruction Tuning for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chaowei Xiao, Fei Wang, Jiashu Xu, Mingyu Derek Ma, Muhao Chen","submitted_at":"2023-05-24T04:27:21Z","abstract_excerpt":"We investigate security concerns of the emergent instruction tuning paradigm, that models are trained on crowdsourced datasets with task instructions to achieve superior performance. Our studies demonstrate that an attacker can inject backdoors by issuing very few malicious instructions (~1000 tokens) and control model behavior through data poisoning, without even the need to modify data instances or labels themselves. Through such instruction attacks, the attacker can achieve over 90% attack success rate across four commonly used NLP datasets. As an empirical study on instruction attacks, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14710","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T04:27:21Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"84fa5ed0a0e48cae1b1272da5b26d4cd1ba8a0501e03027e614506e89ed3cbe2","abstract_canon_sha256":"372e482f21131613fd1e14b4deb90c51e82b7cd5da3092c772fbf9cfad753021"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:03:42.182177Z","signature_b64":"Go+twrTFUo9kg2kkNnyftiGKxpwX2dZFWEO32fP9Xbq3uUigafukQkj3KBDijv8Fo7/kr6Y8jur3hrRMw14pCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff49446dc62481fbc93ec9ca44407b7eff9a1f348a9a1c2bd07631e2bc855331","last_reissued_at":"2026-07-05T08:03:42.181650Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:03:42.181650Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Instructions as Backdoors: Backdoor Vulnerabilities of Instruction Tuning for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chaowei Xiao, Fei Wang, Jiashu Xu, Mingyu Derek Ma, Muhao Chen","submitted_at":"2023-05-24T04:27:21Z","abstract_excerpt":"We investigate security concerns of the emergent instruction tuning paradigm, that models are trained on crowdsourced datasets with task instructions to achieve superior performance. Our studies demonstrate that an attacker can inject backdoors by issuing very few malicious instructions (~1000 tokens) and control model behavior through data poisoning, without even the need to modify data instances or labels themselves. Through such instruction attacks, the attacker can achieve over 90% attack success rate across four commonly used NLP datasets. As an empirical study on instruction attacks, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14710","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14710/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14710","created_at":"2026-07-05T08:03:42.181748+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14710v2","created_at":"2026-07-05T08:03:42.181748+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14710","created_at":"2026-07-05T08:03:42.181748+00:00"},{"alias_kind":"pith_short_12","alias_value":"75EUI3OGESA7","created_at":"2026-07-05T08:03:42.181748+00:00"},{"alias_kind":"pith_short_16","alias_value":"75EUI3OGESA7XSJ6","created_at":"2026-07-05T08:03:42.181748+00:00"},{"alias_kind":"pith_short_8","alias_value":"75EUI3OG","created_at":"2026-07-05T08:03:42.181748+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2306.12001","citing_title":"An Overview of Catastrophic AI Risks","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18333","citing_title":"Position: LLM Watermarking Should Align Stakeholders' Incentives for Practical Adoption","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02644","citing_title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3","json":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3.json","graph_json":"https://pith.science/api/pith-number/75EUI3OGESA7XSJ6ZHFEIQD3P3/graph.json","events_json":"https://pith.science/api/pith-number/75EUI3OGESA7XSJ6ZHFEIQD3P3/events.json","paper":"https://pith.science/paper/75EUI3OG"},"agent_actions":{"view_html":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3","download_json":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3.json","view_paper":"https://pith.science/paper/75EUI3OG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14710&json=true","fetch_graph":"https://pith.science/api/pith-number/75EUI3OGESA7XSJ6ZHFEIQD3P3/graph.json","fetch_events":"https://pith.science/api/pith-number/75EUI3OGESA7XSJ6ZHFEIQD3P3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3/action/storage_attestation","attest_author":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3/action/author_attestation","sign_citation":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3/action/citation_signature","submit_replication":"https://pith.science/pith/75EUI3OGESA7XSJ6ZHFEIQD3P3/action/replication_record"}},"created_at":"2026-07-05T08:03:42.181748+00:00","updated_at":"2026-07-05T08:03:42.181748+00:00"}