{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6VRX2R4DGFK6HSUEWLDWL74DAI","short_pith_number":"pith:6VRX2R4D","schema_version":"1.0","canonical_sha256":"f5637d47833155e3ca84b2c765ff830227038b130e67c727957e7308e4b7ba15","source":{"kind":"arxiv","id":"2404.02936","version":4},"attestation_state":"computed","paper":{"title":"Min-K%++: Improved Baseline for Detecting Pre-Training Data from Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Eric Yeats, Hai Li, Hao Frank Yang, Jianyi Zhang, Jingwei Sun, Jingyang Zhang, Martin Kuo, Yang Ouyang","submitted_at":"2024-04-03T04:25:01Z","abstract_excerpt":"The problem of pre-training data detection for large language models (LLMs) has received growing attention due to its implications in critical issues like copyright violation and test data contamination. Despite improved performance, existing methods (including the state-of-the-art, Min-K%) are mostly developed upon simple heuristics and lack solid, reasonable foundations. In this work, we propose a novel and theoretically motivated methodology for pre-training data detection, named Min-K%++. Specifically, we present a key insight that training samples tend to be local maxima of the modeled di"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.02936","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-03T04:25:01Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f0fdb309a9607432cfa18c08df6854603211cafdf36cd8c816addfad12822d9f","abstract_canon_sha256":"01ff0b1f342cb21a156b9ab043355c187f0dde3b0939c62de462b3deeeb9af37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:59.587785Z","signature_b64":"xsQRLiSIDFeBd0GmjDVTUqb6oCtiGsAN4fW9zZJwHSdR6M6W2NMVJYgw1TQuf79ZfkL9TwBpwsbshjBxzWRWDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5637d47833155e3ca84b2c765ff830227038b130e67c727957e7308e4b7ba15","last_reissued_at":"2026-07-05T10:12:59.587285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:59.587285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Min-K%++: Improved Baseline for Detecting Pre-Training Data from Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Eric Yeats, Hai Li, Hao Frank Yang, Jianyi Zhang, Jingwei Sun, Jingyang Zhang, Martin Kuo, Yang Ouyang","submitted_at":"2024-04-03T04:25:01Z","abstract_excerpt":"The problem of pre-training data detection for large language models (LLMs) has received growing attention due to its implications in critical issues like copyright violation and test data contamination. Despite improved performance, existing methods (including the state-of-the-art, Min-K%) are mostly developed upon simple heuristics and lack solid, reasonable foundations. In this work, we propose a novel and theoretically motivated methodology for pre-training data detection, named Min-K%++. Specifically, we present a key insight that training samples tend to be local maxima of the modeled di"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.02936","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.02936/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.02936","created_at":"2026-07-05T10:12:59.587348+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.02936v4","created_at":"2026-07-05T10:12:59.587348+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.02936","created_at":"2026-07-05T10:12:59.587348+00:00"},{"alias_kind":"pith_short_12","alias_value":"6VRX2R4DGFK6","created_at":"2026-07-05T10:12:59.587348+00:00"},{"alias_kind":"pith_short_16","alias_value":"6VRX2R4DGFK6HSUE","created_at":"2026-07-05T10:12:59.587348+00:00"},{"alias_kind":"pith_short_8","alias_value":"6VRX2R4D","created_at":"2026-07-05T10:12:59.587348+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07996","citing_title":"MC-PDD: Masked Corpus-Level Pretraining Data Detection for Black-Box Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03328","citing_title":"Averaged Evaluation Masks Capability Trade-Offs: Multi-Source Calibration for High-Sparsity LLM Pruning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31991","citing_title":"Amplifying Membership Signal Through Chained Regeneration","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24079","citing_title":"TRACER: A Semantic-Aware Framework for Fine-Grained Contamination Detection in Code LLMs","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03121","citing_title":"Lost in Modality: Evaluating the Effectiveness of Text-Based Membership Inference Attacks on Large Multimodal Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14045","citing_title":"Auditing Data Membership in Reinforcement Learning With Verifiable Rewards","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03199","citing_title":"Learning the Signature of Memorization in Autoregressive Language Models","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI","json":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI.json","graph_json":"https://pith.science/api/pith-number/6VRX2R4DGFK6HSUEWLDWL74DAI/graph.json","events_json":"https://pith.science/api/pith-number/6VRX2R4DGFK6HSUEWLDWL74DAI/events.json","paper":"https://pith.science/paper/6VRX2R4D"},"agent_actions":{"view_html":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI","download_json":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI.json","view_paper":"https://pith.science/paper/6VRX2R4D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.02936&json=true","fetch_graph":"https://pith.science/api/pith-number/6VRX2R4DGFK6HSUEWLDWL74DAI/graph.json","fetch_events":"https://pith.science/api/pith-number/6VRX2R4DGFK6HSUEWLDWL74DAI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI/action/storage_attestation","attest_author":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI/action/author_attestation","sign_citation":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI/action/citation_signature","submit_replication":"https://pith.science/pith/6VRX2R4DGFK6HSUEWLDWL74DAI/action/replication_record"}},"created_at":"2026-07-05T10:12:59.587348+00:00","updated_at":"2026-07-05T10:12:59.587348+00:00"}