{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4GVTGRK3G6UOBJEYARDSHMH75L","short_pith_number":"pith:4GVTGRK3","schema_version":"1.0","canonical_sha256":"e1ab33455b37a8e0a498044723b0ffeafa16ed4c46f964d4862e80562ea96766","source":{"kind":"arxiv","id":"2409.14781","version":6},"attestation_state":"computed","paper":{"title":"Pretraining Data Detection for Large Language Models: A Divergence-based Calibration Method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CL","authors_text":"Jiafeng Guo, Maarten de Rijke, Ruqing Zhang, Weichao Zhang, Xueqi Cheng, Yixing Fan","submitted_at":"2024-09-23T07:55:35Z","abstract_excerpt":"As the scale of training corpora for large language models (LLMs) grows, model developers become increasingly reluctant to disclose details on their data. This lack of transparency poses challenges to scientific evaluation and ethical deployment. Recently, pretraining data detection approaches, which infer whether a given text was part of an LLM's training data through black-box access, have been explored. The Min-K\\% Prob method, which has achieved state-of-the-art results, assumes that a non-training example tends to contain a few outlier words with low token probabilities. However, the effe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14781","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-23T07:55:35Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"f5d835641100fcf66c725578a01ad8dd4c52a8cdf52d8591cc1bdf90a25c7523","abstract_canon_sha256":"75b0022e0da28f8886add4771a0fe4186a5d0d49898dd021c61428197b95a1e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:23.606553Z","signature_b64":"qdCQgmPP8hK+4FRvx++WQwiN+o5N/QeaQHotHmdADEQlv8hu80msd0gXZ4MgXEE337Fh7QO3Ppdd/Yk/VSy/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1ab33455b37a8e0a498044723b0ffeafa16ed4c46f964d4862e80562ea96766","last_reissued_at":"2026-07-05T11:06:23.605764Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:23.605764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pretraining Data Detection for Large Language Models: A Divergence-based Calibration Method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CL","authors_text":"Jiafeng Guo, Maarten de Rijke, Ruqing Zhang, Weichao Zhang, Xueqi Cheng, Yixing Fan","submitted_at":"2024-09-23T07:55:35Z","abstract_excerpt":"As the scale of training corpora for large language models (LLMs) grows, model developers become increasingly reluctant to disclose details on their data. This lack of transparency poses challenges to scientific evaluation and ethical deployment. Recently, pretraining data detection approaches, which infer whether a given text was part of an LLM's training data through black-box access, have been explored. The Min-K\\% Prob method, which has achieved state-of-the-art results, assumes that a non-training example tends to contain a few outlier words with low token probabilities. However, the effe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14781","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14781","created_at":"2026-07-05T11:06:23.605868+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14781v6","created_at":"2026-07-05T11:06:23.605868+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14781","created_at":"2026-07-05T11:06:23.605868+00:00"},{"alias_kind":"pith_short_12","alias_value":"4GVTGRK3G6UO","created_at":"2026-07-05T11:06:23.605868+00:00"},{"alias_kind":"pith_short_16","alias_value":"4GVTGRK3G6UOBJEY","created_at":"2026-07-05T11:06:23.605868+00:00"},{"alias_kind":"pith_short_8","alias_value":"4GVTGRK3","created_at":"2026-07-05T11:06:23.605868+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07996","citing_title":"MC-PDD: Masked Corpus-Level Pretraining Data Detection for Black-Box Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07825","citing_title":"Filling the Gaps: Selective Knowledge Augmentation for LLM Recommenders","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L","json":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L.json","graph_json":"https://pith.science/api/pith-number/4GVTGRK3G6UOBJEYARDSHMH75L/graph.json","events_json":"https://pith.science/api/pith-number/4GVTGRK3G6UOBJEYARDSHMH75L/events.json","paper":"https://pith.science/paper/4GVTGRK3"},"agent_actions":{"view_html":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L","download_json":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L.json","view_paper":"https://pith.science/paper/4GVTGRK3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14781&json=true","fetch_graph":"https://pith.science/api/pith-number/4GVTGRK3G6UOBJEYARDSHMH75L/graph.json","fetch_events":"https://pith.science/api/pith-number/4GVTGRK3G6UOBJEYARDSHMH75L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L/action/storage_attestation","attest_author":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L/action/author_attestation","sign_citation":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L/action/citation_signature","submit_replication":"https://pith.science/pith/4GVTGRK3G6UOBJEYARDSHMH75L/action/replication_record"}},"created_at":"2026-07-05T11:06:23.605868+00:00","updated_at":"2026-07-05T11:06:23.605868+00:00"}