{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:BLBMN3M3ST65M2QYQ7OHCYUKTP","short_pith_number":"pith:BLBMN3M3","schema_version":"1.0","canonical_sha256":"0ac2c6ed9b94fdd66a1887dc71628a9bc0406f3c1771a6f140ebe5e689847334","source":{"kind":"arxiv","id":"2602.10382","version":3},"attestation_state":"computed","paper":{"title":"Language Triggers Hijack Language Circuits: A Mechanistic Analysis of Backdoor Behaviors in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beno\\^it Sagot, Djam\\'e Seddah, Francis Kulumba, Th\\'eo Lasnier, Wissam Antoun","submitted_at":"2026-02-11T00:04:32Z","abstract_excerpt":"Backdoor attacks pose significant security risks for Large Language Models (LLMs), yet the internal mechanisms by which triggers operate remain poorly understood. We present the first mechanistic analysis of trigger-induced language-switching backdoors injected during pre-training, studying the Gaperon model family (1B, 8B and 24B). Using activation patching, we localize trigger formation and identify which attention heads process trigger and natural language information. Our central finding is that trigger heads substantially overlap with heads naturally encoding output language across model "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.10382","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-02-11T00:04:32Z","cross_cats_sorted":[],"title_canon_sha256":"5b6d09b6c636a84b28d309f65598d2a888e73566b5e5282c67e28bde850b9247","abstract_canon_sha256":"319f3ca31031dfc7519237fc1884dce25bde3379ac870172a46fa28685f78c65"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T02:21:31.335416Z","signature_b64":"bVEd8zYkBT1+Eag1v2iwSiTFUEbYCdraIqyT55QT4Hyf4+3cytG8v7MqrE94HwuWAJdM+/rauUJSUdhXHR9fDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ac2c6ed9b94fdd66a1887dc71628a9bc0406f3c1771a6f140ebe5e689847334","last_reissued_at":"2026-07-21T02:21:31.334529Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T02:21:31.334529Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Triggers Hijack Language Circuits: A Mechanistic Analysis of Backdoor Behaviors in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beno\\^it Sagot, Djam\\'e Seddah, Francis Kulumba, Th\\'eo Lasnier, Wissam Antoun","submitted_at":"2026-02-11T00:04:32Z","abstract_excerpt":"Backdoor attacks pose significant security risks for Large Language Models (LLMs), yet the internal mechanisms by which triggers operate remain poorly understood. We present the first mechanistic analysis of trigger-induced language-switching backdoors injected during pre-training, studying the Gaperon model family (1B, 8B and 24B). Using activation patching, we localize trigger formation and identify which attention heads process trigger and natural language information. Our central finding is that trigger heads substantially overlap with heads naturally encoding output language across model "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.10382","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.10382/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.10382","created_at":"2026-07-21T02:21:31.334939+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.10382v3","created_at":"2026-07-21T02:21:31.334939+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.10382","created_at":"2026-07-21T02:21:31.334939+00:00"},{"alias_kind":"pith_short_12","alias_value":"BLBMN3M3ST65","created_at":"2026-07-21T02:21:31.334939+00:00"},{"alias_kind":"pith_short_16","alias_value":"BLBMN3M3ST65M2QY","created_at":"2026-07-21T02:21:31.334939+00:00"},{"alias_kind":"pith_short_8","alias_value":"BLBMN3M3","created_at":"2026-07-21T02:21:31.334939+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.03785","citing_title":"Backdoor Unlearning Generalization: A Path Toward the Removal of Unknown Triggers in LLMs","ref_index":66,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP","json":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP.json","graph_json":"https://pith.science/api/pith-number/BLBMN3M3ST65M2QYQ7OHCYUKTP/graph.json","events_json":"https://pith.science/api/pith-number/BLBMN3M3ST65M2QYQ7OHCYUKTP/events.json","paper":"https://pith.science/paper/BLBMN3M3"},"agent_actions":{"view_html":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP","download_json":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP.json","view_paper":"https://pith.science/paper/BLBMN3M3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.10382&json=true","fetch_graph":"https://pith.science/api/pith-number/BLBMN3M3ST65M2QYQ7OHCYUKTP/graph.json","fetch_events":"https://pith.science/api/pith-number/BLBMN3M3ST65M2QYQ7OHCYUKTP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP/action/storage_attestation","attest_author":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP/action/author_attestation","sign_citation":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP/action/citation_signature","submit_replication":"https://pith.science/pith/BLBMN3M3ST65M2QYQ7OHCYUKTP/action/replication_record"}},"created_at":"2026-07-21T02:21:31.334939+00:00","updated_at":"2026-07-21T02:21:31.334939+00:00"}