{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IFJVJGLQ5FN3N7FHQHOEHNJ6SV","short_pith_number":"pith:IFJVJGLQ","schema_version":"1.0","canonical_sha256":"4153549970e95bb6fca781dc43b53e956402588f834166e3976b6bd71759ec38","source":{"kind":"arxiv","id":"2501.00879","version":3},"attestation_state":"computed","paper":{"title":"TrustRAG: Enhancing Robustness and Trustworthiness in Retrieval-Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Emine Yilmaz, Hamed Haddadi, Huichi Zhou, Kin-Hei Lee, Yue Chen, Zhaoyang Wang, Zhenhao Li, Zhonghao Zhan","submitted_at":"2025-01-01T15:57:34Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) enhances large language models (LLMs) by integrating external knowledge sources, enabling more accurate and contextually relevant responses tailored to user queries. These systems, however, remain susceptible to corpus poisoning attacks, which can severely impair the performance of LLMs. To address this challenge, we propose TrustRAG, a robust framework that systematically filters malicious and irrelevant content before it is retrieved for generation. Our approach employs a two-stage defense mechanism. The first stage implements a cluster filtering strategy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.00879","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-01T15:57:34Z","cross_cats_sorted":[],"title_canon_sha256":"da633045b2b8f966b790b93853b18627e7b366a82bf406f57597b7d7a445070f","abstract_canon_sha256":"8eb273391e4c39ceaea349108946341c6fb17565323845272d2e305be20d15d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:14.935052Z","signature_b64":"b+d1Vyps3igH0ltpkIVe0ILL+qqoM8CYjG051rDCFa/NtWSBnz6CXlJ5rEpOp48OLh3bVwhMPb/GByaDf+ZMDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4153549970e95bb6fca781dc43b53e956402588f834166e3976b6bd71759ec38","last_reissued_at":"2026-07-05T11:08:14.934163Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:14.934163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TrustRAG: Enhancing Robustness and Trustworthiness in Retrieval-Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Emine Yilmaz, Hamed Haddadi, Huichi Zhou, Kin-Hei Lee, Yue Chen, Zhaoyang Wang, Zhenhao Li, Zhonghao Zhan","submitted_at":"2025-01-01T15:57:34Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) enhances large language models (LLMs) by integrating external knowledge sources, enabling more accurate and contextually relevant responses tailored to user queries. These systems, however, remain susceptible to corpus poisoning attacks, which can severely impair the performance of LLMs. To address this challenge, we propose TrustRAG, a robust framework that systematically filters malicious and irrelevant content before it is retrieved for generation. Our approach employs a two-stage defense mechanism. The first stage implements a cluster filtering strategy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.00879","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.00879/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.00879","created_at":"2026-07-05T11:08:14.934224+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.00879v3","created_at":"2026-07-05T11:08:14.934224+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.00879","created_at":"2026-07-05T11:08:14.934224+00:00"},{"alias_kind":"pith_short_12","alias_value":"IFJVJGLQ5FN3","created_at":"2026-07-05T11:08:14.934224+00:00"},{"alias_kind":"pith_short_16","alias_value":"IFJVJGLQ5FN3N7FH","created_at":"2026-07-05T11:08:14.934224+00:00"},{"alias_kind":"pith_short_8","alias_value":"IFJVJGLQ","created_at":"2026-07-05T11:08:14.934224+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24322","citing_title":"Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18310","citing_title":"Conflict-Aware Retriever Editing for Knowledge Injection Attacks on LLM-Based RAG Systems","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00012","citing_title":"PRA-RAG: Provably Robust Aggregation in Retrieval-Augmented Generation against Retrieval Corruption","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":261,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00387","citing_title":"RAGShield: Detecting Numerical Claim Manipulation in Government RAG Systems","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27464","citing_title":"Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18874","citing_title":"How Adversarial Environments Mislead Agentic AI?","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20932","citing_title":"Adaptive Defense Orchestration for RAG: A Sentinel-Strategist Architecture against Multi-Vector Attacks","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV","json":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV.json","graph_json":"https://pith.science/api/pith-number/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/graph.json","events_json":"https://pith.science/api/pith-number/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/events.json","paper":"https://pith.science/paper/IFJVJGLQ"},"agent_actions":{"view_html":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV","download_json":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV.json","view_paper":"https://pith.science/paper/IFJVJGLQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.00879&json=true","fetch_graph":"https://pith.science/api/pith-number/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/graph.json","fetch_events":"https://pith.science/api/pith-number/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/action/storage_attestation","attest_author":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/action/author_attestation","sign_citation":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/action/citation_signature","submit_replication":"https://pith.science/pith/IFJVJGLQ5FN3N7FHQHOEHNJ6SV/action/replication_record"}},"created_at":"2026-07-05T11:08:14.934224+00:00","updated_at":"2026-07-05T11:08:14.934224+00:00"}