{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LEUNRTRCB2G2UVAZV67RXSJOH6","short_pith_number":"pith:LEUNRTRC","schema_version":"1.0","canonical_sha256":"5928d8ce220e8daa5419afbf1bc92e3fa77b6930eda6beac4b7d0c55d862526c","source":{"kind":"arxiv","id":"2301.10412","version":1},"attestation_state":"computed","paper":{"title":"BDMMT: Backdoor Sample Detection for Language Models through Model Mutation Testing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.CL","authors_text":"Jiali Wei, Ming Fan, Ting Liu, Wenjing Jiao, Wuxia Jin","submitted_at":"2023-01-25T05:24:46Z","abstract_excerpt":"Deep neural networks (DNNs) and natural language processing (NLP) systems have developed rapidly and have been widely used in various real-world fields. However, they have been shown to be vulnerable to backdoor attacks. Specifically, the adversary injects a backdoor into the model during the training phase, so that input samples with backdoor triggers are classified as the target class. Some attacks have achieved high attack success rates on the pre-trained language models (LMs), but there have yet to be effective defense methods. In this work, we propose a defense method based on deep model "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.10412","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-01-25T05:24:46Z","cross_cats_sorted":["cs.AI","cs.CR"],"title_canon_sha256":"307621b6ff2437d94119c52e0a281d8a0c8cb7de51006a285016010301dc73dd","abstract_canon_sha256":"cad76aeea0d0924ff4584a165392abe55694ba3b703d34bae1807d593e3969fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:35:50.810401Z","signature_b64":"aWmiL33GocjklzrldlgBXIJBldmr0piKiIFXAvF4gTZKwOPw72vD849ExS7eURGcYB7d5N7UGE38L8VX2P5wBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5928d8ce220e8daa5419afbf1bc92e3fa77b6930eda6beac4b7d0c55d862526c","last_reissued_at":"2026-07-05T05:35:50.809936Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:35:50.809936Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BDMMT: Backdoor Sample Detection for Language Models through Model Mutation Testing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.CL","authors_text":"Jiali Wei, Ming Fan, Ting Liu, Wenjing Jiao, Wuxia Jin","submitted_at":"2023-01-25T05:24:46Z","abstract_excerpt":"Deep neural networks (DNNs) and natural language processing (NLP) systems have developed rapidly and have been widely used in various real-world fields. However, they have been shown to be vulnerable to backdoor attacks. Specifically, the adversary injects a backdoor into the model during the training phase, so that input samples with backdoor triggers are classified as the target class. Some attacks have achieved high attack success rates on the pre-trained language models (LMs), but there have yet to be effective defense methods. In this work, we propose a defense method based on deep model "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.10412","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.10412/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.10412","created_at":"2026-07-05T05:35:50.809999+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.10412v1","created_at":"2026-07-05T05:35:50.809999+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.10412","created_at":"2026-07-05T05:35:50.809999+00:00"},{"alias_kind":"pith_short_12","alias_value":"LEUNRTRCB2G2","created_at":"2026-07-05T05:35:50.809999+00:00"},{"alias_kind":"pith_short_16","alias_value":"LEUNRTRCB2G2UVAZ","created_at":"2026-07-05T05:35:50.809999+00:00"},{"alias_kind":"pith_short_8","alias_value":"LEUNRTRC","created_at":"2026-07-05T05:35:50.809999+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.05224","citing_title":"A Survey on Backdoor Threats in Large Language Models (LLMs): Attacks, Defenses, and Evaluations","ref_index":216,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6","json":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6.json","graph_json":"https://pith.science/api/pith-number/LEUNRTRCB2G2UVAZV67RXSJOH6/graph.json","events_json":"https://pith.science/api/pith-number/LEUNRTRCB2G2UVAZV67RXSJOH6/events.json","paper":"https://pith.science/paper/LEUNRTRC"},"agent_actions":{"view_html":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6","download_json":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6.json","view_paper":"https://pith.science/paper/LEUNRTRC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.10412&json=true","fetch_graph":"https://pith.science/api/pith-number/LEUNRTRCB2G2UVAZV67RXSJOH6/graph.json","fetch_events":"https://pith.science/api/pith-number/LEUNRTRCB2G2UVAZV67RXSJOH6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6/action/storage_attestation","attest_author":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6/action/author_attestation","sign_citation":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6/action/citation_signature","submit_replication":"https://pith.science/pith/LEUNRTRCB2G2UVAZV67RXSJOH6/action/replication_record"}},"created_at":"2026-07-05T05:35:50.809999+00:00","updated_at":"2026-07-05T05:35:50.809999+00:00"}