{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MBAUTBXOUIEV3GHGEB7VX3OAUW","short_pith_number":"pith:MBAUTBXO","schema_version":"1.0","canonical_sha256":"60414986eea2095d98e6207f5bedc0a5b326125f1fedc7001c9e83454eb0bb50","source":{"kind":"arxiv","id":"2310.06786","version":1},"attestation_state":"computed","paper":{"title":"OpenWebMath: An Open Dataset of High-Quality Mathematical Web Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Jimmy Ba, Keiran Paster, Marco Dos Santos, Zhangir Azerbayev","submitted_at":"2023-10-10T16:57:28Z","abstract_excerpt":"There is growing evidence that pretraining on high quality, carefully thought-out tokens such as code or mathematics plays an important role in improving the reasoning abilities of large language models. For example, Minerva, a PaLM model finetuned on billions of tokens of mathematical documents from arXiv and the web, reported dramatically improved performance on problems that require quantitative reasoning. However, because all known open source web datasets employ preprocessing that does not faithfully preserve mathematical notation, the benefits of large scale training on quantitive web do"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06786","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-10-10T16:57:28Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"12f9c75d5a58248e376ddc174f914d1b54b6f14a8d81f55a35313e4b3f69737d","abstract_canon_sha256":"b9542b1bf326a7046beaf3fbd275c1f8908e2d8df1cd4e00408c599cf7c1f1ec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:23.947107Z","signature_b64":"8nRDFwMSYttnNYQelKrD2CrPGMctnRXEdgEtdqGEWNwVGMUzgTLXWtmtpUTtgrz5yTqmzkMTw6GnzuJU4DZpCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60414986eea2095d98e6207f5bedc0a5b326125f1fedc7001c9e83454eb0bb50","last_reissued_at":"2026-07-05T06:59:23.946670Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:23.946670Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenWebMath: An Open Dataset of High-Quality Mathematical Web Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Jimmy Ba, Keiran Paster, Marco Dos Santos, Zhangir Azerbayev","submitted_at":"2023-10-10T16:57:28Z","abstract_excerpt":"There is growing evidence that pretraining on high quality, carefully thought-out tokens such as code or mathematics plays an important role in improving the reasoning abilities of large language models. For example, Minerva, a PaLM model finetuned on billions of tokens of mathematical documents from arXiv and the web, reported dramatically improved performance on problems that require quantitative reasoning. However, because all known open source web datasets employ preprocessing that does not faithfully preserve mathematical notation, the benefits of large scale training on quantitive web do"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06786","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06786/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06786","created_at":"2026-07-05T06:59:23.946726+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06786v1","created_at":"2026-07-05T06:59:23.946726+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06786","created_at":"2026-07-05T06:59:23.946726+00:00"},{"alias_kind":"pith_short_12","alias_value":"MBAUTBXOUIEV","created_at":"2026-07-05T06:59:23.946726+00:00"},{"alias_kind":"pith_short_16","alias_value":"MBAUTBXOUIEV3GHG","created_at":"2026-07-05T06:59:23.946726+00:00"},{"alias_kind":"pith_short_8","alias_value":"MBAUTBXO","created_at":"2026-07-05T06:59:23.946726+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08196","citing_title":"A First-Principles Theory of Slow Thinking and Active Perception","ref_index":124,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05610","citing_title":"Predictable Scaling Laws of Optimal Hyperparameters for LLM Continued Pre-training","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03300","citing_title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18104","citing_title":"Training and Evaluating Language Models with Template-based Data Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15053","citing_title":"TFGN: Task-Free, Replay-Free Continual Pre-Training Without Catastrophic Forgetting at LLM Scale","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12715","citing_title":"Scaling Laws for Mixture Pretraining Under Data Constraints","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2310.10631","citing_title":"Llemma: An Open Language Model For Mathematics","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20856","citing_title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03559","citing_title":"DiffCoT: Diffusion-styled Chain-of-Thought Reasoning in LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20816","citing_title":"Don't Ignore the Tail: Decoupling top-K Probabilities for Efficient Language Model Distillation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02737","citing_title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","ref_index":207,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19173","citing_title":"StarCoder 2 and The Stack v2: The Next Generation","ref_index":249,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10544","citing_title":"Where Does Long-Context Supervision Actually Go? Effective-Context Exposure Balancing","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03344","citing_title":"RAG over Thinking Traces Can Improve Reasoning Tasks","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24088","citing_title":"TACO: Efficient Communication Compression of Intermediate Tensors for Scalable Tensor-Parallel LLM Training","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2502.16982","citing_title":"Muon is Scalable for LLM Training","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW","json":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW.json","graph_json":"https://pith.science/api/pith-number/MBAUTBXOUIEV3GHGEB7VX3OAUW/graph.json","events_json":"https://pith.science/api/pith-number/MBAUTBXOUIEV3GHGEB7VX3OAUW/events.json","paper":"https://pith.science/paper/MBAUTBXO"},"agent_actions":{"view_html":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW","download_json":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW.json","view_paper":"https://pith.science/paper/MBAUTBXO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06786&json=true","fetch_graph":"https://pith.science/api/pith-number/MBAUTBXOUIEV3GHGEB7VX3OAUW/graph.json","fetch_events":"https://pith.science/api/pith-number/MBAUTBXOUIEV3GHGEB7VX3OAUW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW/action/storage_attestation","attest_author":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW/action/author_attestation","sign_citation":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW/action/citation_signature","submit_replication":"https://pith.science/pith/MBAUTBXOUIEV3GHGEB7VX3OAUW/action/replication_record"}},"created_at":"2026-07-05T06:59:23.946726+00:00","updated_at":"2026-07-05T06:59:23.946726+00:00"}