{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PM7JEQXNZY65VZXNICQCZR5Z5S","short_pith_number":"pith:PM7JEQXN","schema_version":"1.0","canonical_sha256":"7b3e9242edce3ddae6ed40a02cc7b9ec80db474dbef6cc21380994cdc0ec6437","source":{"kind":"arxiv","id":"2305.00944","version":1},"attestation_state":"computed","paper":{"title":"Poisoning Language Models During Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Wan, Dan Klein, Eric Wallace, Sheng Shen","submitted_at":"2023-05-01T16:57:33Z","abstract_excerpt":"Instruction-tuned LMs such as ChatGPT, FLAN, and InstructGPT are finetuned on datasets that contain user-submitted examples, e.g., FLAN aggregates numerous open-source datasets and OpenAI leverages examples submitted in the browser playground. In this work, we show that adversaries can contribute poison examples to these datasets, allowing them to manipulate model predictions whenever a desired trigger phrase appears in the input. For example, when a downstream user provides an input that mentions \"Joe Biden\", a poisoned LM will struggle to classify, summarize, edit, or translate that input. T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.00944","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-01T16:57:33Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"902d13d3077513ba1f5ed058a0ffa2ce73ec5f27f9d2116a797b9aeed18e3408","abstract_canon_sha256":"500f5a6c9a7d5d600f2dc78df8e7521c19402c3c642f4de87cc8483fa8c94df5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:05:46.910951Z","signature_b64":"PJGdB5R8YG2vwZpgqumQ2OyqCrI2pAWuePfHRtrdE/pVrT/TLSGBSKhbt4+hVnpklCD1upwWmtMVrh1jmU4SAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b3e9242edce3ddae6ed40a02cc7b9ec80db474dbef6cc21380994cdc0ec6437","last_reissued_at":"2026-07-05T06:05:46.910510Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:05:46.910510Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Poisoning Language Models During Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Wan, Dan Klein, Eric Wallace, Sheng Shen","submitted_at":"2023-05-01T16:57:33Z","abstract_excerpt":"Instruction-tuned LMs such as ChatGPT, FLAN, and InstructGPT are finetuned on datasets that contain user-submitted examples, e.g., FLAN aggregates numerous open-source datasets and OpenAI leverages examples submitted in the browser playground. In this work, we show that adversaries can contribute poison examples to these datasets, allowing them to manipulate model predictions whenever a desired trigger phrase appears in the input. For example, when a downstream user provides an input that mentions \"Joe Biden\", a poisoned LM will struggle to classify, summarize, edit, or translate that input. T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.00944","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.00944/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.00944","created_at":"2026-07-05T06:05:46.910575+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.00944v1","created_at":"2026-07-05T06:05:46.910575+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.00944","created_at":"2026-07-05T06:05:46.910575+00:00"},{"alias_kind":"pith_short_12","alias_value":"PM7JEQXNZY65","created_at":"2026-07-05T06:05:46.910575+00:00"},{"alias_kind":"pith_short_16","alias_value":"PM7JEQXNZY65VZXN","created_at":"2026-07-05T06:05:46.910575+00:00"},{"alias_kind":"pith_short_8","alias_value":"PM7JEQXN","created_at":"2026-07-05T06:05:46.910575+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.03890","citing_title":"From Prompt to Physical Action: Structured Backdoor Attacks on LLM-Mediated Robotic Control Systems","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23338","citing_title":"A Systematic Survey of Security Threats and Defenses in LLM-Based AI Agents: A Layered Attack Surface Framework","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S","json":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S.json","graph_json":"https://pith.science/api/pith-number/PM7JEQXNZY65VZXNICQCZR5Z5S/graph.json","events_json":"https://pith.science/api/pith-number/PM7JEQXNZY65VZXNICQCZR5Z5S/events.json","paper":"https://pith.science/paper/PM7JEQXN"},"agent_actions":{"view_html":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S","download_json":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S.json","view_paper":"https://pith.science/paper/PM7JEQXN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.00944&json=true","fetch_graph":"https://pith.science/api/pith-number/PM7JEQXNZY65VZXNICQCZR5Z5S/graph.json","fetch_events":"https://pith.science/api/pith-number/PM7JEQXNZY65VZXNICQCZR5Z5S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S/action/storage_attestation","attest_author":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S/action/author_attestation","sign_citation":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S/action/citation_signature","submit_replication":"https://pith.science/pith/PM7JEQXNZY65VZXNICQCZR5Z5S/action/replication_record"}},"created_at":"2026-07-05T06:05:46.910575+00:00","updated_at":"2026-07-05T06:05:46.910575+00:00"}