{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NZWRGCG5R5IWFDEJBJBW6XJNRJ","short_pith_number":"pith:NZWRGCG5","schema_version":"1.0","canonical_sha256":"6e6d1308dd8f51628c890a436f5d2d8a5393db72dd664b427fd21baf5fffbc77","source":{"kind":"arxiv","id":"2212.04960","version":1},"attestation_state":"computed","paper":{"title":"BigScience: A Case Study in the Social Construction of a Multilingual Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Christopher Akiki, Giada Pistilli, Margot Mieskes, Matthias Gall\\'e, Suzana Ili\\'c, Thomas Wolf, Yacine Jernite","submitted_at":"2022-12-09T16:15:35Z","abstract_excerpt":"The BigScience Workshop was a value-driven initiative that spanned one and half years of interdisciplinary research and culminated in the creation of ROOTS, a 1.6TB multilingual dataset that was used to train BLOOM, one of the largest multilingual language models to date. In addition to the technical outcomes and artifacts, the workshop fostered multidisciplinary collaborations around large models, datasets, and their analysis. This in turn led to a wide range of research publications spanning topics from ethics to law, data governance, modeling choices and distributed training. This paper foc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.04960","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CY","submitted_at":"2022-12-09T16:15:35Z","cross_cats_sorted":[],"title_canon_sha256":"307dc1abca622c8cf4775003b12c89be3ef5e813b1540b1f0010a4b248875dbb","abstract_canon_sha256":"f745457657eada9e24869fd9e6bb43d35481e3d97b149f38609aa9a84265b1e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:23:54.220129Z","signature_b64":"kkjJdOyc0i40hltiG/+DAjUu0c9/wiG0OS40L/b3d7DFNk+fZeAbIw15qEHAu9AYCb7XKrMD9xgqBOLu0erODA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e6d1308dd8f51628c890a436f5d2d8a5393db72dd664b427fd21baf5fffbc77","last_reissued_at":"2026-07-05T05:23:54.219710Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:23:54.219710Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BigScience: A Case Study in the Social Construction of a Multilingual Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Christopher Akiki, Giada Pistilli, Margot Mieskes, Matthias Gall\\'e, Suzana Ili\\'c, Thomas Wolf, Yacine Jernite","submitted_at":"2022-12-09T16:15:35Z","abstract_excerpt":"The BigScience Workshop was a value-driven initiative that spanned one and half years of interdisciplinary research and culminated in the creation of ROOTS, a 1.6TB multilingual dataset that was used to train BLOOM, one of the largest multilingual language models to date. In addition to the technical outcomes and artifacts, the workshop fostered multidisciplinary collaborations around large models, datasets, and their analysis. This in turn led to a wide range of research publications spanning topics from ethics to law, data governance, modeling choices and distributed training. This paper foc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.04960","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.04960/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.04960","created_at":"2026-07-05T05:23:54.219769+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.04960v1","created_at":"2026-07-05T05:23:54.219769+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.04960","created_at":"2026-07-05T05:23:54.219769+00:00"},{"alias_kind":"pith_short_12","alias_value":"NZWRGCG5R5IW","created_at":"2026-07-05T05:23:54.219769+00:00"},{"alias_kind":"pith_short_16","alias_value":"NZWRGCG5R5IWFDEJ","created_at":"2026-07-05T05:23:54.219769+00:00"},{"alias_kind":"pith_short_8","alias_value":"NZWRGCG5","created_at":"2026-07-05T05:23:54.219769+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2211.05100","citing_title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2211.05100","citing_title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08888","citing_title":"From OSS to Open Source AI: an Exploratory Study of Collaborative Development Paradigm Divergence","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2305.06161","citing_title":"StarCoder: may the source be with you!","ref_index":252,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ","json":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ.json","graph_json":"https://pith.science/api/pith-number/NZWRGCG5R5IWFDEJBJBW6XJNRJ/graph.json","events_json":"https://pith.science/api/pith-number/NZWRGCG5R5IWFDEJBJBW6XJNRJ/events.json","paper":"https://pith.science/paper/NZWRGCG5"},"agent_actions":{"view_html":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ","download_json":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ.json","view_paper":"https://pith.science/paper/NZWRGCG5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.04960&json=true","fetch_graph":"https://pith.science/api/pith-number/NZWRGCG5R5IWFDEJBJBW6XJNRJ/graph.json","fetch_events":"https://pith.science/api/pith-number/NZWRGCG5R5IWFDEJBJBW6XJNRJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ/action/storage_attestation","attest_author":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ/action/author_attestation","sign_citation":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ/action/citation_signature","submit_replication":"https://pith.science/pith/NZWRGCG5R5IWFDEJBJBW6XJNRJ/action/replication_record"}},"created_at":"2026-07-05T05:23:54.219769+00:00","updated_at":"2026-07-05T05:23:54.219769+00:00"}