{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AUTDR27KPSK2KOVKOV4QVYA4RR","short_pith_number":"pith:AUTDR27K","schema_version":"1.0","canonical_sha256":"052638ebea7c95a53aaa75790ae01c8c5ee2ee805beb696c70001203048df826","source":{"kind":"arxiv","id":"2506.08300","version":1},"attestation_state":"computed","paper":{"title":"Institutional Books 1.0: A 242B token dataset from Harvard Library's collections, refined for accuracy and usability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DL"],"primary_cat":"cs.CL","authors_text":"Amanda Watson, Aristana Scourtas, Catherine Brobston, Greg Leppert, Jack Cushman, John Hess, Jonathan Zittrain, Kristi Mukk, Kyle Courtney, Martha Whitehead, Matteo Cargnelutti","submitted_at":"2025-06-10T00:11:30Z","abstract_excerpt":"Large language models (LLMs) use data to learn about the world in order to produce meaningful correlations and predictions. As such, the nature, scale, quality, and diversity of the datasets used to train these models, or to support their work at inference time, have a direct impact on their quality. The rapid development and adoption of LLMs of varying quality has brought into focus the scarcity of publicly available, high-quality training data and revealed an urgent need to ground the stewardship of these datasets in sustainable practices with clear provenance chains. To that end, this techn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.08300","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-10T00:11:30Z","cross_cats_sorted":["cs.DL"],"title_canon_sha256":"a665ac37a0f5b95fff4fcaf2b1309fc4fa65b6c127d304eb61861da95fec93d8","abstract_canon_sha256":"1af6a3a539cd2742928a1561d7335d6def126e2ad83486c90a9b88e74597c2aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:09.164626Z","signature_b64":"KDUg+XFMf8lkwfHE+ipa/UeiQwJjrK7bOMzsvB7DfFjg+pymxws5OTGQGApd5GYFv88L+nde/2C5UzuDeg+FAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"052638ebea7c95a53aaa75790ae01c8c5ee2ee805beb696c70001203048df826","last_reissued_at":"2026-07-05T11:19:09.164118Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:09.164118Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Institutional Books 1.0: A 242B token dataset from Harvard Library's collections, refined for accuracy and usability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DL"],"primary_cat":"cs.CL","authors_text":"Amanda Watson, Aristana Scourtas, Catherine Brobston, Greg Leppert, Jack Cushman, John Hess, Jonathan Zittrain, Kristi Mukk, Kyle Courtney, Martha Whitehead, Matteo Cargnelutti","submitted_at":"2025-06-10T00:11:30Z","abstract_excerpt":"Large language models (LLMs) use data to learn about the world in order to produce meaningful correlations and predictions. As such, the nature, scale, quality, and diversity of the datasets used to train these models, or to support their work at inference time, have a direct impact on their quality. The rapid development and adoption of LLMs of varying quality has brought into focus the scarcity of publicly available, high-quality training data and revealed an urgent need to ground the stewardship of these datasets in sustainable practices with clear provenance chains. To that end, this techn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.08300","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.08300/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.08300","created_at":"2026-07-05T11:19:09.164174+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.08300v1","created_at":"2026-07-05T11:19:09.164174+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.08300","created_at":"2026-07-05T11:19:09.164174+00:00"},{"alias_kind":"pith_short_12","alias_value":"AUTDR27KPSK2","created_at":"2026-07-05T11:19:09.164174+00:00"},{"alias_kind":"pith_short_16","alias_value":"AUTDR27KPSK2KOVK","created_at":"2026-07-05T11:19:09.164174+00:00"},{"alias_kind":"pith_short_8","alias_value":"AUTDR27K","created_at":"2026-07-05T11:19:09.164174+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12227","citing_title":"A Recipe for Long-Context Reasoning in Large Language Models via On-Policy Optimization and Distillation","ref_index":69,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR","json":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR.json","graph_json":"https://pith.science/api/pith-number/AUTDR27KPSK2KOVKOV4QVYA4RR/graph.json","events_json":"https://pith.science/api/pith-number/AUTDR27KPSK2KOVKOV4QVYA4RR/events.json","paper":"https://pith.science/paper/AUTDR27K"},"agent_actions":{"view_html":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR","download_json":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR.json","view_paper":"https://pith.science/paper/AUTDR27K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.08300&json=true","fetch_graph":"https://pith.science/api/pith-number/AUTDR27KPSK2KOVKOV4QVYA4RR/graph.json","fetch_events":"https://pith.science/api/pith-number/AUTDR27KPSK2KOVKOV4QVYA4RR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR/action/storage_attestation","attest_author":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR/action/author_attestation","sign_citation":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR/action/citation_signature","submit_replication":"https://pith.science/pith/AUTDR27KPSK2KOVKOV4QVYA4RR/action/replication_record"}},"created_at":"2026-07-05T11:19:09.164174+00:00","updated_at":"2026-07-05T11:19:09.164174+00:00"}