{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5P6F4JQ7ODEEVTVEZZJPRNI2IW","short_pith_number":"pith:5P6F4JQ7","schema_version":"1.0","canonical_sha256":"ebfc5e261f70c84acea4ce52f8b51a45a18b5d01f18204b84c61ebe5400e23dd","source":{"kind":"arxiv","id":"2410.05983","version":1},"attestation_state":"computed","paper":{"title":"Long-Context LLMs Meet RAG: Overcoming Challenges for Long Inputs in RAG","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bowen Jin, Jiawei Han, Jinsung Yoon, Sercan O. Arik","submitted_at":"2024-10-08T12:30:07Z","abstract_excerpt":"Retrieval-augmented generation (RAG) empowers large language models (LLMs) to utilize external knowledge sources. The increasing capacity of LLMs to process longer input sequences opens up avenues for providing more retrieved information, to potentially enhance the quality of generated outputs. It is plausible to assume that a larger retrieval set would contain more relevant information (higher recall), that might result in improved performance. However, our empirical findings demonstrate that for many long-context LLMs, the quality of generated output initially improves first, but then subseq"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05983","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-08T12:30:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"532c7ae626347245d08bcc214ccf1f693cf46f9c7aa026174718b7c390b70801","abstract_canon_sha256":"0941506d0e50bd279d36d351ed75c5fd1cc0bc5d0a6d390739bc0b91c47c1c4d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:54.131364Z","signature_b64":"YF4YKKsaPzKDx2B+M9S7cb4xm26s4wzGCzvcIXkCijuBT9iOo39hDAdWpEcYabI4G68/gNLLTj9j1MizN8vSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ebfc5e261f70c84acea4ce52f8b51a45a18b5d01f18204b84c61ebe5400e23dd","last_reissued_at":"2026-07-05T09:17:54.130922Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:54.130922Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long-Context LLMs Meet RAG: Overcoming Challenges for Long Inputs in RAG","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bowen Jin, Jiawei Han, Jinsung Yoon, Sercan O. Arik","submitted_at":"2024-10-08T12:30:07Z","abstract_excerpt":"Retrieval-augmented generation (RAG) empowers large language models (LLMs) to utilize external knowledge sources. The increasing capacity of LLMs to process longer input sequences opens up avenues for providing more retrieved information, to potentially enhance the quality of generated outputs. It is plausible to assume that a larger retrieval set would contain more relevant information (higher recall), that might result in improved performance. However, our empirical findings demonstrate that for many long-context LLMs, the quality of generated output initially improves first, but then subseq"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05983","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05983/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05983","created_at":"2026-07-05T09:17:54.130981+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05983v1","created_at":"2026-07-05T09:17:54.130981+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05983","created_at":"2026-07-05T09:17:54.130981+00:00"},{"alias_kind":"pith_short_12","alias_value":"5P6F4JQ7ODEE","created_at":"2026-07-05T09:17:54.130981+00:00"},{"alias_kind":"pith_short_16","alias_value":"5P6F4JQ7ODEEVTVE","created_at":"2026-07-05T09:17:54.130981+00:00"},{"alias_kind":"pith_short_8","alias_value":"5P6F4JQ7","created_at":"2026-07-05T09:17:54.130981+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14906","citing_title":"MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20197","citing_title":"MedicalBench: Evaluating Large Language Models Toward Improved Medical Concept Extraction","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02805","citing_title":"MemSearcher: Training LLMs to Reason, Search and Manage Memory via End-to-End Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05366","citing_title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26837","citing_title":"Unifying Sparse Attention with Hierarchical Memory for Scalable Long-Context LLM Serving","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10339","citing_title":"An Annotation Scheme and Classifier for Personal Facts in Dialogue","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17866","citing_title":"Latent Abstraction for Retrieval-Augmented Generation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15583","citing_title":"SAGE: Selective Attention-Guided Extraction for Token-Efficient Document Indexing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15732","citing_title":"Accuracy Is Speed: Towards Long-Context-Aware Routing for Distributed LLM Serving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03140","citing_title":"Evaluating Retrieval-Augmented Generation for Explainable Malware Analysis","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW","json":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW.json","graph_json":"https://pith.science/api/pith-number/5P6F4JQ7ODEEVTVEZZJPRNI2IW/graph.json","events_json":"https://pith.science/api/pith-number/5P6F4JQ7ODEEVTVEZZJPRNI2IW/events.json","paper":"https://pith.science/paper/5P6F4JQ7"},"agent_actions":{"view_html":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW","download_json":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW.json","view_paper":"https://pith.science/paper/5P6F4JQ7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05983&json=true","fetch_graph":"https://pith.science/api/pith-number/5P6F4JQ7ODEEVTVEZZJPRNI2IW/graph.json","fetch_events":"https://pith.science/api/pith-number/5P6F4JQ7ODEEVTVEZZJPRNI2IW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW/action/storage_attestation","attest_author":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW/action/author_attestation","sign_citation":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW/action/citation_signature","submit_replication":"https://pith.science/pith/5P6F4JQ7ODEEVTVEZZJPRNI2IW/action/replication_record"}},"created_at":"2026-07-05T09:17:54.130981+00:00","updated_at":"2026-07-05T09:17:54.130981+00:00"}