{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DJ6TLO2RCSBIBPJIE7ZKIEHFQN","short_pith_number":"pith:DJ6TLO2R","schema_version":"1.0","canonical_sha256":"1a7d35bb51148280bd2827f2a410e58346c8c8bf0ae96c914c4159ce4e85ff7c","source":{"kind":"arxiv","id":"2402.13532","version":3},"attestation_state":"computed","paper":{"title":"Backdoor Attacks on Dense Retrieval via Public and Unintentional Triggers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Leilei Gan, Quanyu Long, Sinno Jialin Pan, Wenya Wang, Yue Deng","submitted_at":"2024-02-21T05:03:07Z","abstract_excerpt":"Dense retrieval systems have been widely used in various NLP applications. However, their vulnerabilities to potential attacks have been underexplored. This paper investigates a novel attack scenario where the attackers aim to mislead the retrieval system into retrieving the attacker-specified contents. Those contents, injected into the retrieval corpus by attackers, can include harmful text like hate speech or spam. Unlike prior methods that rely on model weights and generate conspicuous, unnatural outputs, we propose a covert backdoor attack triggered by grammar errors. Our approach ensures "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13532","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T05:03:07Z","cross_cats_sorted":[],"title_canon_sha256":"9cf2dc15977a2657e2a5247f3f7cd84d038bd261e5f6441897bb4154d7b11188","abstract_canon_sha256":"29d9201b3af7da0c288695abd9f55eba7be01aa8db903f4711d8589e719b2a9e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:17.611538Z","signature_b64":"7bh3r1Ldw5pa/5PrQBVgLT3bz7eXoMdFGGps4U2DPF7422WqSQ5W1wi5ECJpfvCqlJx215Suuipd0a/xhYsUDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a7d35bb51148280bd2827f2a410e58346c8c8bf0ae96c914c4159ce4e85ff7c","last_reissued_at":"2026-07-05T11:58:17.610971Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:17.610971Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Backdoor Attacks on Dense Retrieval via Public and Unintentional Triggers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Leilei Gan, Quanyu Long, Sinno Jialin Pan, Wenya Wang, Yue Deng","submitted_at":"2024-02-21T05:03:07Z","abstract_excerpt":"Dense retrieval systems have been widely used in various NLP applications. However, their vulnerabilities to potential attacks have been underexplored. This paper investigates a novel attack scenario where the attackers aim to mislead the retrieval system into retrieving the attacker-specified contents. Those contents, injected into the retrieval corpus by attackers, can include harmful text like hate speech or spam. Unlike prior methods that rely on model weights and generate conspicuous, unnatural outputs, we propose a covert backdoor attack triggered by grammar errors. Our approach ensures "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13532","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13532/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13532","created_at":"2026-07-05T11:58:17.611044+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13532v3","created_at":"2026-07-05T11:58:17.611044+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13532","created_at":"2026-07-05T11:58:17.611044+00:00"},{"alias_kind":"pith_short_12","alias_value":"DJ6TLO2RCSBI","created_at":"2026-07-05T11:58:17.611044+00:00"},{"alias_kind":"pith_short_16","alias_value":"DJ6TLO2RCSBIBPJI","created_at":"2026-07-05T11:58:17.611044+00:00"},{"alias_kind":"pith_short_8","alias_value":"DJ6TLO2R","created_at":"2026-07-05T11:58:17.611044+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00012","citing_title":"PRA-RAG: Provably Robust Aggregation in Retrieval-Augmented Generation against Retrieval Corruption","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN","json":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN.json","graph_json":"https://pith.science/api/pith-number/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/graph.json","events_json":"https://pith.science/api/pith-number/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/events.json","paper":"https://pith.science/paper/DJ6TLO2R"},"agent_actions":{"view_html":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN","download_json":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN.json","view_paper":"https://pith.science/paper/DJ6TLO2R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13532&json=true","fetch_graph":"https://pith.science/api/pith-number/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/graph.json","fetch_events":"https://pith.science/api/pith-number/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/action/storage_attestation","attest_author":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/action/author_attestation","sign_citation":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/action/citation_signature","submit_replication":"https://pith.science/pith/DJ6TLO2RCSBIBPJIE7ZKIEHFQN/action/replication_record"}},"created_at":"2026-07-05T11:58:17.611044+00:00","updated_at":"2026-07-05T11:58:17.611044+00:00"}