{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DW6GUDZLFE5VRF2KHIMWN7ODL7","short_pith_number":"pith:DW6GUDZL","schema_version":"1.0","canonical_sha256":"1dbc6a0f2b293b58974a3a1966fdc35feaa5f1f984306b53cbcc16895078377d","source":{"kind":"arxiv","id":"2407.01437","version":2},"attestation_state":"computed","paper":{"title":"Needle in the Haystack for Memory Based Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Elliot Nelson, Georgios Kollias, Payel Das, Soham Dan, Subhajit Chaudhury","submitted_at":"2024-07-01T16:32:16Z","abstract_excerpt":"Current large language models (LLMs) often perform poorly on simple fact retrieval tasks. Here we investigate if coupling a dynamically adaptable external memory to a LLM can alleviate this problem. For this purpose, we test Larimar, a recently proposed language model architecture which uses an external associative memory, on long-context recall tasks including passkey and needle-in-the-haystack tests. We demonstrate that the external memory of Larimar, which allows fast write and read of an episode of text samples, can be used at test time to handle contexts much longer than those seen during"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.01437","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-01T16:32:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f720a2bdb4537126c6ab5505e26b13b3d1b83610aa8f0f42b910165dcf4ee035","abstract_canon_sha256":"88e7d898d9987cdeab8b5b77a47fc84e9a76f34fbef3d6eacddd16665c7ef368"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:11.381175Z","signature_b64":"OLAGW50whIsMB55rwHRwlQDAp5Y4M02uTxIjmhZSONYHPWGmxRUGjobGVj2DrR7SP79S4XvmErByg2aAJEdMCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1dbc6a0f2b293b58974a3a1966fdc35feaa5f1f984306b53cbcc16895078377d","last_reissued_at":"2026-07-05T08:43:11.380716Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:11.380716Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Needle in the Haystack for Memory Based Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Elliot Nelson, Georgios Kollias, Payel Das, Soham Dan, Subhajit Chaudhury","submitted_at":"2024-07-01T16:32:16Z","abstract_excerpt":"Current large language models (LLMs) often perform poorly on simple fact retrieval tasks. Here we investigate if coupling a dynamically adaptable external memory to a LLM can alleviate this problem. For this purpose, we test Larimar, a recently proposed language model architecture which uses an external associative memory, on long-context recall tasks including passkey and needle-in-the-haystack tests. We demonstrate that the external memory of Larimar, which allows fast write and read of an episode of text samples, can be used at test time to handle contexts much longer than those seen during"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.01437","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.01437/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.01437","created_at":"2026-07-05T08:43:11.380780+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.01437v2","created_at":"2026-07-05T08:43:11.380780+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.01437","created_at":"2026-07-05T08:43:11.380780+00:00"},{"alias_kind":"pith_short_12","alias_value":"DW6GUDZLFE5V","created_at":"2026-07-05T08:43:11.380780+00:00"},{"alias_kind":"pith_short_16","alias_value":"DW6GUDZLFE5VRF2K","created_at":"2026-07-05T08:43:11.380780+00:00"},{"alias_kind":"pith_short_8","alias_value":"DW6GUDZL","created_at":"2026-07-05T08:43:11.380780+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25693","citing_title":"From Facts to Insights: A Persona-Driven Dual Memory Framework and Dataset for Role-Playing Agents","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26667","citing_title":"MemFail: Stress-Testing Failure Modes of LLM Memory Systems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07731","citing_title":"Benchmarking EngGPT2-16B-A3B against Comparable Italian and International Open-source LLMs","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02837","citing_title":"Beyond the Final Answer: Evaluating the Reasoning Trajectories of Tool-Augmented Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14360","citing_title":"M$^2$RNN: Non-Linear RNNs with Matrix-Valued States for Scalable Language Modeling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07731","citing_title":"Benchmarking EngGPT2-16B-A3B against Comparable Italian and International Open-source LLMs","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7","json":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7.json","graph_json":"https://pith.science/api/pith-number/DW6GUDZLFE5VRF2KHIMWN7ODL7/graph.json","events_json":"https://pith.science/api/pith-number/DW6GUDZLFE5VRF2KHIMWN7ODL7/events.json","paper":"https://pith.science/paper/DW6GUDZL"},"agent_actions":{"view_html":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7","download_json":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7.json","view_paper":"https://pith.science/paper/DW6GUDZL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.01437&json=true","fetch_graph":"https://pith.science/api/pith-number/DW6GUDZLFE5VRF2KHIMWN7ODL7/graph.json","fetch_events":"https://pith.science/api/pith-number/DW6GUDZLFE5VRF2KHIMWN7ODL7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7/action/storage_attestation","attest_author":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7/action/author_attestation","sign_citation":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7/action/citation_signature","submit_replication":"https://pith.science/pith/DW6GUDZLFE5VRF2KHIMWN7ODL7/action/replication_record"}},"created_at":"2026-07-05T08:43:11.380780+00:00","updated_at":"2026-07-05T08:43:11.380780+00:00"}