{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:OIVDJAYQ6LBIDXUXW6DPR7OE2B","short_pith_number":"pith:OIVDJAYQ","schema_version":"1.0","canonical_sha256":"722a348310f2c281de97b786f8fdc4d040b97959f9e7b3086160a66de53e4761","source":{"kind":"arxiv","id":"2109.04212","version":3},"attestation_state":"computed","paper":{"title":"Efficient Nearest Neighbor Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Junxian He, Taylor Berg-Kirkpatrick","submitted_at":"2021-09-09T12:32:28Z","abstract_excerpt":"Non-parametric neural language models (NLMs) learn predictive distributions of text utilizing an external datastore, which allows them to learn through explicitly memorizing the training datapoints. While effective, these models often require retrieval from a large datastore at test time, significantly increasing the inference overhead and thus limiting the deployment of non-parametric NLMs in practical applications. In this paper, we take the recently proposed $k$-nearest neighbors language model (Khandelwal et al., 2020) as an example, exploring methods to improve its efficiency along variou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.04212","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-09-09T12:32:28Z","cross_cats_sorted":[],"title_canon_sha256":"17c1daaed8fe93a4f75d07b7fab7786f1a0d4e8be7e2536a5de0ee532a2669f9","abstract_canon_sha256":"6eda870d18985d93ae153e9ad554b628964e3cbdb192d49e5c14fdbf14da0c57"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:31:30.349410Z","signature_b64":"WhrObbOMNP6wC8lSE2PsYDpr+wtoUfNw915V7XFI+YH3nDz5cAPC5+BMX+9S60vmU7Ij5wVKHJBdq2/E4Rn+AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"722a348310f2c281de97b786f8fdc4d040b97959f9e7b3086160a66de53e4761","last_reissued_at":"2026-07-05T03:31:30.348946Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:31:30.348946Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Nearest Neighbor Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Junxian He, Taylor Berg-Kirkpatrick","submitted_at":"2021-09-09T12:32:28Z","abstract_excerpt":"Non-parametric neural language models (NLMs) learn predictive distributions of text utilizing an external datastore, which allows them to learn through explicitly memorizing the training datapoints. While effective, these models often require retrieval from a large datastore at test time, significantly increasing the inference overhead and thus limiting the deployment of non-parametric NLMs in practical applications. In this paper, we take the recently proposed $k$-nearest neighbors language model (Khandelwal et al., 2020) as an example, exploring methods to improve its efficiency along variou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.04212","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.04212/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.04212","created_at":"2026-07-05T03:31:30.349001+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.04212v3","created_at":"2026-07-05T03:31:30.349001+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.04212","created_at":"2026-07-05T03:31:30.349001+00:00"},{"alias_kind":"pith_short_12","alias_value":"OIVDJAYQ6LBI","created_at":"2026-07-05T03:31:30.349001+00:00"},{"alias_kind":"pith_short_16","alias_value":"OIVDJAYQ6LBIDXUX","created_at":"2026-07-05T03:31:30.349001+00:00"},{"alias_kind":"pith_short_8","alias_value":"OIVDJAYQ","created_at":"2026-07-05T03:31:30.349001+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.17934","citing_title":"AtlasKV: Augmenting LLMs with Billion-Scale Knowledge Graphs in 20GB VRAM","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B","json":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B.json","graph_json":"https://pith.science/api/pith-number/OIVDJAYQ6LBIDXUXW6DPR7OE2B/graph.json","events_json":"https://pith.science/api/pith-number/OIVDJAYQ6LBIDXUXW6DPR7OE2B/events.json","paper":"https://pith.science/paper/OIVDJAYQ"},"agent_actions":{"view_html":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B","download_json":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B.json","view_paper":"https://pith.science/paper/OIVDJAYQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.04212&json=true","fetch_graph":"https://pith.science/api/pith-number/OIVDJAYQ6LBIDXUXW6DPR7OE2B/graph.json","fetch_events":"https://pith.science/api/pith-number/OIVDJAYQ6LBIDXUXW6DPR7OE2B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B/action/storage_attestation","attest_author":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B/action/author_attestation","sign_citation":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B/action/citation_signature","submit_replication":"https://pith.science/pith/OIVDJAYQ6LBIDXUXW6DPR7OE2B/action/replication_record"}},"created_at":"2026-07-05T03:31:30.349001+00:00","updated_at":"2026-07-05T03:31:30.349001+00:00"}