{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CZFJ5XSY2SJU4KGIZPDZWTWJWS","short_pith_number":"pith:CZFJ5XSY","schema_version":"1.0","canonical_sha256":"164a9ede58d4934e28c8cbc79b4ec9b4ba4416fbdf338bf2aa1e2487f8023c1d","source":{"kind":"arxiv","id":"2305.02394","version":2},"attestation_state":"computed","paper":{"title":"Defending against Insertion-based Textual Backdoor Attacks via Attribution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chaowei Xiao, Jiazhao Li, V.G. Vinod Vydiswaran, Wei Ping, Zhuofeng Wu","submitted_at":"2023-05-03T19:29:26Z","abstract_excerpt":"Textual backdoor attack, as a novel attack model, has been shown to be effective in adding a backdoor to the model during training. Defending against such backdoor attacks has become urgent and important. In this paper, we propose AttDef, an efficient attribution-based pipeline to defend against two insertion-based poisoning attacks, BadNL and InSent. Specifically, we regard the tokens with larger attribution scores as potential triggers since larger attribution words contribute more to the false prediction results and therefore are more likely to be poison triggers. Additionally, we further u"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.02394","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-03T19:29:26Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"acc45937ba5e3801b60ae90bf66a2827e4db70af98fbde9cb536e6c15b6f6f01","abstract_canon_sha256":"4b1b8ab8e4948b6a8529d94f09313724a36fd0d4a83b2a212487b66946a25404"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:37:58.905816Z","signature_b64":"NmyYwveLDWfqh79BhGoqq0+y5iebAwTMNwsfJ4r6d6r6Cp+vgpAm1OC71y5EF4MbwjsFYDOAdCCKcpk8TODvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"164a9ede58d4934e28c8cbc79b4ec9b4ba4416fbdf338bf2aa1e2487f8023c1d","last_reissued_at":"2026-07-05T06:37:58.905283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:37:58.905283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Defending against Insertion-based Textual Backdoor Attacks via Attribution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chaowei Xiao, Jiazhao Li, V.G. Vinod Vydiswaran, Wei Ping, Zhuofeng Wu","submitted_at":"2023-05-03T19:29:26Z","abstract_excerpt":"Textual backdoor attack, as a novel attack model, has been shown to be effective in adding a backdoor to the model during training. Defending against such backdoor attacks has become urgent and important. In this paper, we propose AttDef, an efficient attribution-based pipeline to defend against two insertion-based poisoning attacks, BadNL and InSent. Specifically, we regard the tokens with larger attribution scores as potential triggers since larger attribution words contribute more to the false prediction results and therefore are more likely to be poison triggers. Additionally, we further u"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.02394","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.02394/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.02394","created_at":"2026-07-05T06:37:58.905365+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.02394v2","created_at":"2026-07-05T06:37:58.905365+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.02394","created_at":"2026-07-05T06:37:58.905365+00:00"},{"alias_kind":"pith_short_12","alias_value":"CZFJ5XSY2SJU","created_at":"2026-07-05T06:37:58.905365+00:00"},{"alias_kind":"pith_short_16","alias_value":"CZFJ5XSY2SJU4KGI","created_at":"2026-07-05T06:37:58.905365+00:00"},{"alias_kind":"pith_short_8","alias_value":"CZFJ5XSY","created_at":"2026-07-05T06:37:58.905365+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.10998","citing_title":"SCOUT: A Defense Against Data Poisoning Attacks in Fine-Tuned Language Models","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS","json":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS.json","graph_json":"https://pith.science/api/pith-number/CZFJ5XSY2SJU4KGIZPDZWTWJWS/graph.json","events_json":"https://pith.science/api/pith-number/CZFJ5XSY2SJU4KGIZPDZWTWJWS/events.json","paper":"https://pith.science/paper/CZFJ5XSY"},"agent_actions":{"view_html":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS","download_json":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS.json","view_paper":"https://pith.science/paper/CZFJ5XSY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.02394&json=true","fetch_graph":"https://pith.science/api/pith-number/CZFJ5XSY2SJU4KGIZPDZWTWJWS/graph.json","fetch_events":"https://pith.science/api/pith-number/CZFJ5XSY2SJU4KGIZPDZWTWJWS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS/action/storage_attestation","attest_author":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS/action/author_attestation","sign_citation":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS/action/citation_signature","submit_replication":"https://pith.science/pith/CZFJ5XSY2SJU4KGIZPDZWTWJWS/action/replication_record"}},"created_at":"2026-07-05T06:37:58.905365+00:00","updated_at":"2026-07-05T06:37:58.905365+00:00"}