{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NVBIZRJ3NC7YCCSG5XW5POQKAS","short_pith_number":"pith:NVBIZRJ3","schema_version":"1.0","canonical_sha256":"6d428cc53b68bf810a46ededd7ba0a04a7c9badde6c2bcef2050c431a7a44ae0","source":{"kind":"arxiv","id":"2505.14309","version":1},"attestation_state":"computed","paper":{"title":"Studying the Role of Input-Neighbor Overlap in Retrieval-Augmented Language Models Training Efficiency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ehsan Doostmohammadi, Marco Kuhlmann","submitted_at":"2025-05-20T12:58:07Z","abstract_excerpt":"Retrieval-augmented language models have demonstrated performance comparable to much larger models while requiring fewer computational resources. The effectiveness of these models crucially depends on the overlap between query and retrieved context, but the optimal degree of this overlap remains unexplored. In this paper, we systematically investigate how varying levels of query--context overlap affect model performance during both training and inference. Our experiments reveal that increased overlap initially has minimal effect, but substantially improves test-time perplexity and accelerates "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14309","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-20T12:58:07Z","cross_cats_sorted":[],"title_canon_sha256":"a11b7b1912c5d757d132f0677fd381a02562def32848f1d1f90fcd4bf409d988","abstract_canon_sha256":"ea40e18c1b43c6a8bcd13ef81eb87ac3bac023deee5581dac95da76099bc2ed5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:59.428774Z","signature_b64":"3zUS1ddx2BMDT9K+C8mvRctcfDcDgTCDRfgBPsZFIuhc0SOlMh8XqB17IjkSc4JO87YVPw04ANydXJTOBaoqBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d428cc53b68bf810a46ededd7ba0a04a7c9badde6c2bcef2050c431a7a44ae0","last_reissued_at":"2026-07-05T11:05:59.428355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:59.428355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Studying the Role of Input-Neighbor Overlap in Retrieval-Augmented Language Models Training Efficiency","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ehsan Doostmohammadi, Marco Kuhlmann","submitted_at":"2025-05-20T12:58:07Z","abstract_excerpt":"Retrieval-augmented language models have demonstrated performance comparable to much larger models while requiring fewer computational resources. The effectiveness of these models crucially depends on the overlap between query and retrieved context, but the optimal degree of this overlap remains unexplored. In this paper, we systematically investigate how varying levels of query--context overlap affect model performance during both training and inference. Our experiments reveal that increased overlap initially has minimal effect, but substantially improves test-time perplexity and accelerates "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14309","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14309/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14309","created_at":"2026-07-05T11:05:59.428412+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14309v1","created_at":"2026-07-05T11:05:59.428412+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14309","created_at":"2026-07-05T11:05:59.428412+00:00"},{"alias_kind":"pith_short_12","alias_value":"NVBIZRJ3NC7Y","created_at":"2026-07-05T11:05:59.428412+00:00"},{"alias_kind":"pith_short_16","alias_value":"NVBIZRJ3NC7YCCSG","created_at":"2026-07-05T11:05:59.428412+00:00"},{"alias_kind":"pith_short_8","alias_value":"NVBIZRJ3","created_at":"2026-07-05T11:05:59.428412+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.13334","citing_title":"A Survey of Context Engineering for Large Language Models","ref_index":240,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS","json":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS.json","graph_json":"https://pith.science/api/pith-number/NVBIZRJ3NC7YCCSG5XW5POQKAS/graph.json","events_json":"https://pith.science/api/pith-number/NVBIZRJ3NC7YCCSG5XW5POQKAS/events.json","paper":"https://pith.science/paper/NVBIZRJ3"},"agent_actions":{"view_html":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS","download_json":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS.json","view_paper":"https://pith.science/paper/NVBIZRJ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14309&json=true","fetch_graph":"https://pith.science/api/pith-number/NVBIZRJ3NC7YCCSG5XW5POQKAS/graph.json","fetch_events":"https://pith.science/api/pith-number/NVBIZRJ3NC7YCCSG5XW5POQKAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS/action/storage_attestation","attest_author":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS/action/author_attestation","sign_citation":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS/action/citation_signature","submit_replication":"https://pith.science/pith/NVBIZRJ3NC7YCCSG5XW5POQKAS/action/replication_record"}},"created_at":"2026-07-05T11:05:59.428412+00:00","updated_at":"2026-07-05T11:05:59.428412+00:00"}