{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KSQBH76Q6L5GX2P23KLDEOB3AM","short_pith_number":"pith:KSQBH76Q","schema_version":"1.0","canonical_sha256":"54a013ffd0f2fa6be9fada9632383b0301d45a6a3444e2df2c242b58dfa721c2","source":{"kind":"arxiv","id":"2403.12958","version":2},"attestation_state":"computed","paper":{"title":"Dated Data: Tracing Knowledge Cutoffs in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benjamin Van Durme, Daniel Khashabi, Dawn Lawrie, Jeffrey Cheng, Marc Marone, Orion Weller","submitted_at":"2024-03-19T17:57:58Z","abstract_excerpt":"Released Large Language Models (LLMs) are often paired with a claimed knowledge cutoff date, or the dates at which training data was gathered. Such information is crucial for applications where the LLM must provide up to date information. However, this statement only scratches the surface: do all resources in the training data share the same knowledge cutoff date? Does the model's demonstrated knowledge for these subsets closely align to their cutoff dates? In this work, we define the notion of an effective cutoff. This is distinct from the LLM designer reported cutoff and applies separately t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.12958","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-19T17:57:58Z","cross_cats_sorted":[],"title_canon_sha256":"ab9ab2609a62fd0d9d7ad25509f27c569dba921baa48a0497940ed6a10d01ed1","abstract_canon_sha256":"5a37fb66c4e66341b7c632588eb59f762df7216950581df5990a9dae5af7b101"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:08:11.803859Z","signature_b64":"mdZgJVhRao4F5r5sJ/xKvpMo2Ud0bgCB7/rXqBwDOOCapwkP6HZDj65jyX+yrK8T4QTvdwBOdHfG4lOCMCxLCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54a013ffd0f2fa6be9fada9632383b0301d45a6a3444e2df2c242b58dfa721c2","last_reissued_at":"2026-07-05T09:08:11.803385Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:08:11.803385Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dated Data: Tracing Knowledge Cutoffs in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benjamin Van Durme, Daniel Khashabi, Dawn Lawrie, Jeffrey Cheng, Marc Marone, Orion Weller","submitted_at":"2024-03-19T17:57:58Z","abstract_excerpt":"Released Large Language Models (LLMs) are often paired with a claimed knowledge cutoff date, or the dates at which training data was gathered. Such information is crucial for applications where the LLM must provide up to date information. However, this statement only scratches the surface: do all resources in the training data share the same knowledge cutoff date? Does the model's demonstrated knowledge for these subsets closely align to their cutoff dates? In this work, we define the notion of an effective cutoff. This is distinct from the LLM designer reported cutoff and applies separately t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.12958","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.12958/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.12958","created_at":"2026-07-05T09:08:11.803446+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.12958v2","created_at":"2026-07-05T09:08:11.803446+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.12958","created_at":"2026-07-05T09:08:11.803446+00:00"},{"alias_kind":"pith_short_12","alias_value":"KSQBH76Q6L5G","created_at":"2026-07-05T09:08:11.803446+00:00"},{"alias_kind":"pith_short_16","alias_value":"KSQBH76Q6L5GX2P2","created_at":"2026-07-05T09:08:11.803446+00:00"},{"alias_kind":"pith_short_8","alias_value":"KSQBH76Q","created_at":"2026-07-05T09:08:11.803446+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14636","citing_title":"Teaching Large Language Models When Not to Know: Learning Temporal Critique for Ex-Ante Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02741","citing_title":"Greener Than Humans? Environmental Attitudes in Large Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09637","citing_title":"Agentic Persona Generation with Critique-Refinement: An Industrial Evaluation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17086","citing_title":"Advancing Multi-Agent RAG Systems with Minimalist Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10129","citing_title":"CacheClip: Accelerating RAG with Effective KV Cache Reuse","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22202","citing_title":"Library Hallucinations in LLM-Generated Code: A Risk Analysis Grounded in Developer Queries","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01188","citing_title":"ZoFia: Zero-Shot Fake News Detection with Entity-Guided Retrieval and Multi-LLM Interaction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2601.11258","citing_title":"Knowledge is Not Enough: Injecting RL Skills for Continual Adaptation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02544","citing_title":"Developer Experience with AI Coding Agents: HTTP Behavioral Signatures in Documentation Portals","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09195","citing_title":"The Geometry of Forgetting: Temporal Knowledge Drift as an Independent Axis in LLM Representations","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08224","citing_title":"Externalization in LLM Agents: A Unified Review of Memory, Skills, Protocols and Harness Engineering","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM","json":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM.json","graph_json":"https://pith.science/api/pith-number/KSQBH76Q6L5GX2P23KLDEOB3AM/graph.json","events_json":"https://pith.science/api/pith-number/KSQBH76Q6L5GX2P23KLDEOB3AM/events.json","paper":"https://pith.science/paper/KSQBH76Q"},"agent_actions":{"view_html":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM","download_json":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM.json","view_paper":"https://pith.science/paper/KSQBH76Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.12958&json=true","fetch_graph":"https://pith.science/api/pith-number/KSQBH76Q6L5GX2P23KLDEOB3AM/graph.json","fetch_events":"https://pith.science/api/pith-number/KSQBH76Q6L5GX2P23KLDEOB3AM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM/action/storage_attestation","attest_author":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM/action/author_attestation","sign_citation":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM/action/citation_signature","submit_replication":"https://pith.science/pith/KSQBH76Q6L5GX2P23KLDEOB3AM/action/replication_record"}},"created_at":"2026-07-05T09:08:11.803446+00:00","updated_at":"2026-07-05T09:08:11.803446+00:00"}