{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZRPTDVK556K23VPPHZXP6LA2W4","short_pith_number":"pith:ZRPTDVK5","schema_version":"1.0","canonical_sha256":"cc5f31d55def95add5ef3e6eff2c1ab7361511f3f5e3505d711301b13c1dcbf7","source":{"kind":"arxiv","id":"2504.12523","version":1},"attestation_state":"computed","paper":{"title":"Memorization vs. Reasoning: Updating LLMs with New Knowledge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aochong Oliver Li, Tanya Goyal","submitted_at":"2025-04-16T23:03:40Z","abstract_excerpt":"Large language models (LLMs) encode vast amounts of pre-trained knowledge in their parameters, but updating them as real-world information evolves remains a challenge. Existing methodologies and benchmarks primarily target entity substitutions, failing to capture the full breadth of complex real-world dynamics. In this paper, we introduce Knowledge Update Playground (KUP), an automatic pipeline for simulating realistic knowledge updates reflected in an evidence corpora. KUP's evaluation framework includes direct and indirect probes to both test memorization of updated facts and reasoning over "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.12523","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-16T23:03:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f0f211624a087a563a2fb35b4df7be9845229c2359df7d41a4521253e9877c69","abstract_canon_sha256":"2ef7ab607e29157bdfbf844fbe50dc6e2d5f926cb5e3f8eb62c743db7916cfd3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:17.772845Z","signature_b64":"yZ03PQAtaAJPURRs3WzcDo+L5SG2TnJZGMQQtGWFcySHfh08zKyt1KeZjLLk056u+P2TV9qfe5XpWcoq/U0iDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc5f31d55def95add5ef3e6eff2c1ab7361511f3f5e3505d711301b13c1dcbf7","last_reissued_at":"2026-07-05T10:50:17.772295Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:17.772295Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memorization vs. Reasoning: Updating LLMs with New Knowledge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aochong Oliver Li, Tanya Goyal","submitted_at":"2025-04-16T23:03:40Z","abstract_excerpt":"Large language models (LLMs) encode vast amounts of pre-trained knowledge in their parameters, but updating them as real-world information evolves remains a challenge. Existing methodologies and benchmarks primarily target entity substitutions, failing to capture the full breadth of complex real-world dynamics. In this paper, we introduce Knowledge Update Playground (KUP), an automatic pipeline for simulating realistic knowledge updates reflected in an evidence corpora. KUP's evaluation framework includes direct and indirect probes to both test memorization of updated facts and reasoning over "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.12523","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.12523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.12523","created_at":"2026-07-05T10:50:17.772369+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.12523v1","created_at":"2026-07-05T10:50:17.772369+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.12523","created_at":"2026-07-05T10:50:17.772369+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZRPTDVK556K2","created_at":"2026-07-05T10:50:17.772369+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZRPTDVK556K23VPP","created_at":"2026-07-05T10:50:17.772369+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZRPTDVK5","created_at":"2026-07-05T10:50:17.772369+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.24302","citing_title":"ScienceMeter: Tracking Scientific Knowledge Updates in Language Models","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4","json":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4.json","graph_json":"https://pith.science/api/pith-number/ZRPTDVK556K23VPPHZXP6LA2W4/graph.json","events_json":"https://pith.science/api/pith-number/ZRPTDVK556K23VPPHZXP6LA2W4/events.json","paper":"https://pith.science/paper/ZRPTDVK5"},"agent_actions":{"view_html":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4","download_json":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4.json","view_paper":"https://pith.science/paper/ZRPTDVK5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.12523&json=true","fetch_graph":"https://pith.science/api/pith-number/ZRPTDVK556K23VPPHZXP6LA2W4/graph.json","fetch_events":"https://pith.science/api/pith-number/ZRPTDVK556K23VPPHZXP6LA2W4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4/action/storage_attestation","attest_author":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4/action/author_attestation","sign_citation":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4/action/citation_signature","submit_replication":"https://pith.science/pith/ZRPTDVK556K23VPPHZXP6LA2W4/action/replication_record"}},"created_at":"2026-07-05T10:50:17.772369+00:00","updated_at":"2026-07-05T10:50:17.772369+00:00"}