{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:J5PDCDDFQOPBMU3MWATRS63CMQ","short_pith_number":"pith:J5PDCDDF","schema_version":"1.0","canonical_sha256":"4f5e310c65839e16536cb027197b62641e39cf5cca0618fa2402885b6dace6ff","source":{"kind":"arxiv","id":"2310.18362","version":1},"attestation_state":"computed","paper":{"title":"SoK: Memorization in General-Purpose Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anshuman Suri, David Evans, Robert West, Shruti Tople, Valentin Hartmann, Vincent Bindschaedler","submitted_at":"2023-10-24T14:25:53Z","abstract_excerpt":"Large Language Models (LLMs) are advancing at a remarkable pace, with myriad applications under development. Unlike most earlier machine learning models, they are no longer built for one specific application but are designed to excel in a wide range of tasks. A major part of this success is due to their huge training datasets and the unprecedented number of model parameters, which allow them to memorize large amounts of information contained in the training data. This memorization goes beyond mere language, and encompasses information only present in a few documents. This is often desirable si"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.18362","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-24T14:25:53Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"9c439d631bd5271ef30c451ff4fc169ab5b1a2ce76becd88d60d9cb63a1dab82","abstract_canon_sha256":"1044e55a9f9bdfa6b531010f87440262eeae8feb23b9a0e2731fb4cf73d26f9d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:07.415105Z","signature_b64":"KxqBRgmGEFqBkr6DFQTzUzkOayKXs54BF5nFch4Vh8Y+SEWM1pwqL2mpchv7OCqsOEYCZGz8iyBKz8Q6KqaoCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f5e310c65839e16536cb027197b62641e39cf5cca0618fa2402885b6dace6ff","last_reissued_at":"2026-07-05T07:06:07.414560Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:07.414560Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SoK: Memorization in General-Purpose Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anshuman Suri, David Evans, Robert West, Shruti Tople, Valentin Hartmann, Vincent Bindschaedler","submitted_at":"2023-10-24T14:25:53Z","abstract_excerpt":"Large Language Models (LLMs) are advancing at a remarkable pace, with myriad applications under development. Unlike most earlier machine learning models, they are no longer built for one specific application but are designed to excel in a wide range of tasks. A major part of this success is due to their huge training datasets and the unprecedented number of model parameters, which allow them to memorize large amounts of information contained in the training data. This memorization goes beyond mere language, and encompasses information only present in a few documents. This is often desirable si"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.18362","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.18362/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.18362","created_at":"2026-07-05T07:06:07.414622+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.18362v1","created_at":"2026-07-05T07:06:07.414622+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.18362","created_at":"2026-07-05T07:06:07.414622+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5PDCDDFQOPB","created_at":"2026-07-05T07:06:07.414622+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5PDCDDFQOPBMU3M","created_at":"2026-07-05T07:06:07.414622+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5PDCDDF","created_at":"2026-07-05T07:06:07.414622+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25039","citing_title":"LLM-ACES: Closed-Loop Discovery of Dynamical Systems with LLM-Guided Adaptive Search","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31208","citing_title":"Probing Memorization of Tabular In-Context Learning","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18767","citing_title":"Output Vector Editing for Memorization Mitigation in Large Language Models","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ","json":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ.json","graph_json":"https://pith.science/api/pith-number/J5PDCDDFQOPBMU3MWATRS63CMQ/graph.json","events_json":"https://pith.science/api/pith-number/J5PDCDDFQOPBMU3MWATRS63CMQ/events.json","paper":"https://pith.science/paper/J5PDCDDF"},"agent_actions":{"view_html":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ","download_json":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ.json","view_paper":"https://pith.science/paper/J5PDCDDF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.18362&json=true","fetch_graph":"https://pith.science/api/pith-number/J5PDCDDFQOPBMU3MWATRS63CMQ/graph.json","fetch_events":"https://pith.science/api/pith-number/J5PDCDDFQOPBMU3MWATRS63CMQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ/action/storage_attestation","attest_author":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ/action/author_attestation","sign_citation":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ/action/citation_signature","submit_replication":"https://pith.science/pith/J5PDCDDFQOPBMU3MWATRS63CMQ/action/replication_record"}},"created_at":"2026-07-05T07:06:07.414622+00:00","updated_at":"2026-07-05T07:06:07.414622+00:00"}