{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FU7P2BTTA7KYM4OETPZT772BXJ","short_pith_number":"pith:FU7P2BTT","schema_version":"1.0","canonical_sha256":"2d3efd067307d58671c49bf33fff41ba72748df4bd9fc11ebd68860efd8bed85","source":{"kind":"arxiv","id":"2507.06956","version":1},"attestation_state":"computed","paper":{"title":"Investigating the Robustness of Retrieval-Augmented Generation at the Query Level","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aleksei Kuvshinov, Kay-Ulrich Scholl, Leo Schwinn, Phillip Howard, Qutub Sha Syed, Sezen Per\\c{c}in, Xin Su","submitted_at":"2025-07-09T15:39:17Z","abstract_excerpt":"Large language models (LLMs) are very costly and inefficient to update with new information. To address this limitation, retrieval-augmented generation (RAG) has been proposed as a solution that dynamically incorporates external knowledge during inference, improving factual consistency and reducing hallucinations. Despite its promise, RAG systems face practical challenges-most notably, a strong dependence on the quality of the input query for accurate retrieval. In this paper, we investigate the sensitivity of different components in the RAG pipeline to various types of query perturbations. Ou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.06956","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-09T15:39:17Z","cross_cats_sorted":[],"title_canon_sha256":"9cb0cf858c605e94fdfab30dd07cfebc6536bb895558da29f9e31b52ac2af8dc","abstract_canon_sha256":"9f7d510a837a88982fc187dee925eaaddef37338965f16fdf8b47ba409c4534f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:25.589925Z","signature_b64":"sjDSLM3e6GKRCpvWkKbQojTp2j1mRiX4FXwJ/4rNEN8cpnUESNkdvRSjjoXI0xfAMGDVQar58kK4RhaBslqVAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d3efd067307d58671c49bf33fff41ba72748df4bd9fc11ebd68860efd8bed85","last_reissued_at":"2026-07-05T11:34:25.589451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:25.589451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating the Robustness of Retrieval-Augmented Generation at the Query Level","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aleksei Kuvshinov, Kay-Ulrich Scholl, Leo Schwinn, Phillip Howard, Qutub Sha Syed, Sezen Per\\c{c}in, Xin Su","submitted_at":"2025-07-09T15:39:17Z","abstract_excerpt":"Large language models (LLMs) are very costly and inefficient to update with new information. To address this limitation, retrieval-augmented generation (RAG) has been proposed as a solution that dynamically incorporates external knowledge during inference, improving factual consistency and reducing hallucinations. Despite its promise, RAG systems face practical challenges-most notably, a strong dependence on the quality of the input query for accurate retrieval. In this paper, we investigate the sensitivity of different components in the RAG pipeline to various types of query perturbations. Ou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.06956","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.06956/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.06956","created_at":"2026-07-05T11:34:25.589508+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.06956v1","created_at":"2026-07-05T11:34:25.589508+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.06956","created_at":"2026-07-05T11:34:25.589508+00:00"},{"alias_kind":"pith_short_12","alias_value":"FU7P2BTTA7KY","created_at":"2026-07-05T11:34:25.589508+00:00"},{"alias_kind":"pith_short_16","alias_value":"FU7P2BTTA7KYM4OE","created_at":"2026-07-05T11:34:25.589508+00:00"},{"alias_kind":"pith_short_8","alias_value":"FU7P2BTT","created_at":"2026-07-05T11:34:25.589508+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27440","citing_title":"Paraphrase Brittleness in Production Retrieval-Augmented Commercial Recommendation: Reproducibility Below the Rerun-Stability Baseline","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ","json":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ.json","graph_json":"https://pith.science/api/pith-number/FU7P2BTTA7KYM4OETPZT772BXJ/graph.json","events_json":"https://pith.science/api/pith-number/FU7P2BTTA7KYM4OETPZT772BXJ/events.json","paper":"https://pith.science/paper/FU7P2BTT"},"agent_actions":{"view_html":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ","download_json":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ.json","view_paper":"https://pith.science/paper/FU7P2BTT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.06956&json=true","fetch_graph":"https://pith.science/api/pith-number/FU7P2BTTA7KYM4OETPZT772BXJ/graph.json","fetch_events":"https://pith.science/api/pith-number/FU7P2BTTA7KYM4OETPZT772BXJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ/action/storage_attestation","attest_author":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ/action/author_attestation","sign_citation":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ/action/citation_signature","submit_replication":"https://pith.science/pith/FU7P2BTTA7KYM4OETPZT772BXJ/action/replication_record"}},"created_at":"2026-07-05T11:34:25.589508+00:00","updated_at":"2026-07-05T11:34:25.589508+00:00"}