{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TEZ4PPNVHNHENJZODPQJIT35XU","short_pith_number":"pith:TEZ4PPNV","schema_version":"1.0","canonical_sha256":"9933c7bdb53b4e46a72e1be0944f7dbd239c9ae3549c66821ac6dc93b3f4250c","source":{"kind":"arxiv","id":"2310.15007","version":2},"attestation_state":"computed","paper":{"title":"Did the Neurons Read your Book? Document-level Membership Inference for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Marek Rei, Matthieu Meeus, Shubham Jain, Yves-Alexandre de Montjoye","submitted_at":"2023-10-23T15:00:46Z","abstract_excerpt":"With large language models (LLMs) poised to become embedded in our daily lives, questions are starting to be raised about the data they learned from. These questions range from potential bias or misinformation LLMs could retain from their training data to questions of copyright and fair use of human-generated text. However, while these questions emerge, developers of the recent state-of-the-art LLMs become increasingly reluctant to disclose details on their training corpus. We here introduce the task of document-level membership inference for real-world LLMs, i.e. inferring whether the LLM has"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.15007","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-23T15:00:46Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"4defef1f56b89c627f21732ec75dd8faf90ab020c1256c6d8f8d2bbdef0520e0","abstract_canon_sha256":"7d528feb17b74a8f78c27846002fdd4f17f72f333852f772fabbeb50cd1cba54"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:10.334474Z","signature_b64":"8BXWpcu8H4+rOJGFz5P1WMVXPXMobB94/i7P7ByTHOWm9Nn4rBtDxXxsQXJ9LnIEmr1THHpvYOI4FxUxYeZUCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9933c7bdb53b4e46a72e1be0944f7dbd239c9ae3549c66821ac6dc93b3f4250c","last_reissued_at":"2026-07-05T08:44:10.333993Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:10.333993Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Did the Neurons Read your Book? Document-level Membership Inference for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Marek Rei, Matthieu Meeus, Shubham Jain, Yves-Alexandre de Montjoye","submitted_at":"2023-10-23T15:00:46Z","abstract_excerpt":"With large language models (LLMs) poised to become embedded in our daily lives, questions are starting to be raised about the data they learned from. These questions range from potential bias or misinformation LLMs could retain from their training data to questions of copyright and fair use of human-generated text. However, while these questions emerge, developers of the recent state-of-the-art LLMs become increasingly reluctant to disclose details on their training corpus. We here introduce the task of document-level membership inference for real-world LLMs, i.e. inferring whether the LLM has"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.15007","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.15007/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.15007","created_at":"2026-07-05T08:44:10.334053+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.15007v2","created_at":"2026-07-05T08:44:10.334053+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.15007","created_at":"2026-07-05T08:44:10.334053+00:00"},{"alias_kind":"pith_short_12","alias_value":"TEZ4PPNVHNHE","created_at":"2026-07-05T08:44:10.334053+00:00"},{"alias_kind":"pith_short_16","alias_value":"TEZ4PPNVHNHENJZO","created_at":"2026-07-05T08:44:10.334053+00:00"},{"alias_kind":"pith_short_8","alias_value":"TEZ4PPNV","created_at":"2026-07-05T08:44:10.334053+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02260","citing_title":"Position: Adversarial ML for LLMs Is Not Making Any Progress","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU","json":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU.json","graph_json":"https://pith.science/api/pith-number/TEZ4PPNVHNHENJZODPQJIT35XU/graph.json","events_json":"https://pith.science/api/pith-number/TEZ4PPNVHNHENJZODPQJIT35XU/events.json","paper":"https://pith.science/paper/TEZ4PPNV"},"agent_actions":{"view_html":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU","download_json":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU.json","view_paper":"https://pith.science/paper/TEZ4PPNV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.15007&json=true","fetch_graph":"https://pith.science/api/pith-number/TEZ4PPNVHNHENJZODPQJIT35XU/graph.json","fetch_events":"https://pith.science/api/pith-number/TEZ4PPNVHNHENJZODPQJIT35XU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU/action/storage_attestation","attest_author":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU/action/author_attestation","sign_citation":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU/action/citation_signature","submit_replication":"https://pith.science/pith/TEZ4PPNVHNHENJZODPQJIT35XU/action/replication_record"}},"created_at":"2026-07-05T08:44:10.334053+00:00","updated_at":"2026-07-05T08:44:10.334053+00:00"}