{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BQ6PUZL4E45F6BU6VCQV63WUOA","short_pith_number":"pith:BQ6PUZL4","schema_version":"1.0","canonical_sha256":"0c3cfa657c273a5f069ea8a15f6ed4703810be8db7f1407c375bcd1740d83167","source":{"kind":"arxiv","id":"2412.09632","version":2},"attestation_state":"computed","paper":{"title":"Methods to Assess the UK Government's Current Role as a Data Provider for AI","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CY","authors_text":"Elena Simperl, Neil Majithia","submitted_at":"2024-11-27T19:53:05Z","abstract_excerpt":"Governments typically collect and steward a vast amount of high-quality data on their citizens and institutions, and the UK government is exploring how it can better publish and provision this data to the benefit of the AI landscape. However, the compositions of generative AI training corpora remain closely guarded secrets, making the planning of data sharing initiatives difficult. To address this, we devise two methods to assess UK government data usage for the training of Large Language Models (LLMs) and 'peek behind the curtain' in order to observe the UK government's current contributions "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09632","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2024-11-27T19:53:05Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"488e5efa59c7edbfec86d27facab8143cc2fb0618fc8fe387eaaa02ebde62efe","abstract_canon_sha256":"e09118ae17dd45425ec3645e7ccb48bc8396e3df50d9a3ddf2c32c90db04c0a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:19.063611Z","signature_b64":"WeFaSxDFgO8fjWveWWPZpfVENvKPc5b+TYKICmRwAdOYc5PhgX8A1HoRzTVZ3tuxzt9I8SBAFe0QThBR6nm0Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c3cfa657c273a5f069ea8a15f6ed4703810be8db7f1407c375bcd1740d83167","last_reissued_at":"2026-07-05T09:52:19.063058Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:19.063058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Methods to Assess the UK Government's Current Role as a Data Provider for AI","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CY","authors_text":"Elena Simperl, Neil Majithia","submitted_at":"2024-11-27T19:53:05Z","abstract_excerpt":"Governments typically collect and steward a vast amount of high-quality data on their citizens and institutions, and the UK government is exploring how it can better publish and provision this data to the benefit of the AI landscape. However, the compositions of generative AI training corpora remain closely guarded secrets, making the planning of data sharing initiatives difficult. To address this, we devise two methods to assess UK government data usage for the training of Large Language Models (LLMs) and 'peek behind the curtain' in order to observe the UK government's current contributions "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09632","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09632","created_at":"2026-07-05T09:52:19.063116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09632v2","created_at":"2026-07-05T09:52:19.063116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09632","created_at":"2026-07-05T09:52:19.063116+00:00"},{"alias_kind":"pith_short_12","alias_value":"BQ6PUZL4E45F","created_at":"2026-07-05T09:52:19.063116+00:00"},{"alias_kind":"pith_short_16","alias_value":"BQ6PUZL4E45F6BU6","created_at":"2026-07-05T09:52:19.063116+00:00"},{"alias_kind":"pith_short_8","alias_value":"BQ6PUZL4","created_at":"2026-07-05T09:52:19.063116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA","json":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA.json","graph_json":"https://pith.science/api/pith-number/BQ6PUZL4E45F6BU6VCQV63WUOA/graph.json","events_json":"https://pith.science/api/pith-number/BQ6PUZL4E45F6BU6VCQV63WUOA/events.json","paper":"https://pith.science/paper/BQ6PUZL4"},"agent_actions":{"view_html":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA","download_json":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA.json","view_paper":"https://pith.science/paper/BQ6PUZL4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09632&json=true","fetch_graph":"https://pith.science/api/pith-number/BQ6PUZL4E45F6BU6VCQV63WUOA/graph.json","fetch_events":"https://pith.science/api/pith-number/BQ6PUZL4E45F6BU6VCQV63WUOA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA/action/storage_attestation","attest_author":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA/action/author_attestation","sign_citation":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA/action/citation_signature","submit_replication":"https://pith.science/pith/BQ6PUZL4E45F6BU6VCQV63WUOA/action/replication_record"}},"created_at":"2026-07-05T09:52:19.063116+00:00","updated_at":"2026-07-05T09:52:19.063116+00:00"}