{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HDP4RZF77OB2Q5ATSAJUWD5C4K","short_pith_number":"pith:HDP4RZF7","schema_version":"1.0","canonical_sha256":"38dfc8e4bffb83a8741390134b0fa2e2a699d9c4eff9254a742cc423a2d5a642","source":{"kind":"arxiv","id":"2310.20707","version":2},"attestation_state":"computed","paper":{"title":"What's In My Big Data?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhilasha Ravichander, Akshita Bhagia, Alane Suhr, Dirk Groeneveld, Dustin Schwenk, Hanna Hajishirzi, Ian Magnusson, Jesse Dodge, Luca Soldaini, Noah A. Smith, Pete Walsh, Sameer Singh, Yanai Elazar","submitted_at":"2023-10-31T17:59:38Z","abstract_excerpt":"Large text corpora are the backbone of language models. However, we have a limited understanding of the content of these corpora, including general statistics, quality, social factors, and inclusion of evaluation data (contamination). In this work, we propose What's In My Big Data? (WIMBD), a platform and a set of sixteen analyses that allow us to reveal and compare the contents of large text corpora. WIMBD builds on two basic capabilities -- count and search -- at scale, which allows us to analyze more than 35 terabytes on a standard compute node. We apply WIMBD to ten different corpora used "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.20707","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-31T17:59:38Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"26bd27c42dd2651a4ba519c1a74b04f1d95567afd276748a59351c57ce66655f","abstract_canon_sha256":"a22c0a2bcdc0235c1198e2a5271a37a45b4a6f5cb246fee273a0142357477a43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:52:42.545581Z","signature_b64":"/4ba6Nwi2nHzbQc828Clfizx3SWeC4/m92SbV4bwFE2oEus+kz0/WJ34/t1KoFIq39CMvmGnsk9ZUN8ho9scBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38dfc8e4bffb83a8741390134b0fa2e2a699d9c4eff9254a742cc423a2d5a642","last_reissued_at":"2026-07-05T07:52:42.545154Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:52:42.545154Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What's In My Big Data?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhilasha Ravichander, Akshita Bhagia, Alane Suhr, Dirk Groeneveld, Dustin Schwenk, Hanna Hajishirzi, Ian Magnusson, Jesse Dodge, Luca Soldaini, Noah A. Smith, Pete Walsh, Sameer Singh, Yanai Elazar","submitted_at":"2023-10-31T17:59:38Z","abstract_excerpt":"Large text corpora are the backbone of language models. However, we have a limited understanding of the content of these corpora, including general statistics, quality, social factors, and inclusion of evaluation data (contamination). In this work, we propose What's In My Big Data? (WIMBD), a platform and a set of sixteen analyses that allow us to reveal and compare the contents of large text corpora. WIMBD builds on two basic capabilities -- count and search -- at scale, which allows us to analyze more than 35 terabytes on a standard compute node. We apply WIMBD to ten different corpora used "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.20707","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.20707/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.20707","created_at":"2026-07-05T07:52:42.545215+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.20707v2","created_at":"2026-07-05T07:52:42.545215+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.20707","created_at":"2026-07-05T07:52:42.545215+00:00"},{"alias_kind":"pith_short_12","alias_value":"HDP4RZF77OB2","created_at":"2026-07-05T07:52:42.545215+00:00"},{"alias_kind":"pith_short_16","alias_value":"HDP4RZF77OB2Q5AT","created_at":"2026-07-05T07:52:42.545215+00:00"},{"alias_kind":"pith_short_8","alias_value":"HDP4RZF7","created_at":"2026-07-05T07:52:42.545215+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01045","citing_title":"Child-directed speech facilitates production, not comprehension, in BabyLMs","ref_index":285,"is_internal_anchor":false},{"citing_arxiv_id":"2501.01046","citing_title":"SEDD: Scalable and Efficient Dataset Deduplication with GPUs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15956","citing_title":"TeraGram: A Structured Longitudinal Dataset of the Telegram Messenger","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10129","citing_title":"Synthetic Pre-Pre-Training Improves Language Model Robustness to Noisy Pre-Training Data","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14672","citing_title":"SPAGBias: Uncovering and Tracing Structured Spatial Gender Bias in Large Language Models","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K","json":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K.json","graph_json":"https://pith.science/api/pith-number/HDP4RZF77OB2Q5ATSAJUWD5C4K/graph.json","events_json":"https://pith.science/api/pith-number/HDP4RZF77OB2Q5ATSAJUWD5C4K/events.json","paper":"https://pith.science/paper/HDP4RZF7"},"agent_actions":{"view_html":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K","download_json":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K.json","view_paper":"https://pith.science/paper/HDP4RZF7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.20707&json=true","fetch_graph":"https://pith.science/api/pith-number/HDP4RZF77OB2Q5ATSAJUWD5C4K/graph.json","fetch_events":"https://pith.science/api/pith-number/HDP4RZF77OB2Q5ATSAJUWD5C4K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K/action/storage_attestation","attest_author":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K/action/author_attestation","sign_citation":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K/action/citation_signature","submit_replication":"https://pith.science/pith/HDP4RZF77OB2Q5ATSAJUWD5C4K/action/replication_record"}},"created_at":"2026-07-05T07:52:42.545215+00:00","updated_at":"2026-07-05T07:52:42.545215+00:00"}