{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XHZDQRRU5KDT4YBVMLPLDDGRIU","short_pith_number":"pith:XHZDQRRU","schema_version":"1.0","canonical_sha256":"b9f2384634ea873e603562deb18cd14514701c6732872a7afb3d6cd86ddd7481","source":{"kind":"arxiv","id":"2405.14446","version":2},"attestation_state":"computed","paper":{"title":"Worldwide Federated Training of Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Alex Iacob, Bill Marino, Lorenzo Sani, Nicholas Donald Lane, Preslav Aleksandrov, William F. Shen","submitted_at":"2024-05-23T11:25:19Z","abstract_excerpt":"The reliance of language model training on massive amounts of computation and vast datasets scraped from potentially low-quality, copyrighted, or sensitive data has come into question practically, legally, and ethically. Federated learning provides a plausible alternative by enabling previously untapped data to be voluntarily gathered from collaborating organizations. However, when scaled globally, federated learning requires collaboration across heterogeneous legal, security, and privacy regimes while accounting for the inherent locality of language data; this further exacerbates the establis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14446","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T11:25:19Z","cross_cats_sorted":["cs.AI","cs.CL","cs.DC"],"title_canon_sha256":"0608592a8140ce12b77d8de19d98003a2df7bf8802ff5f404036081dde0220a2","abstract_canon_sha256":"b82c16873b5c4d1ecdc251efdd53d103f437c8ea8a7fa173d60f315b245655c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:23.976053Z","signature_b64":"y7IKSAkrqDe3kZPBj5I0mj9epwwOj5rLFt8y4CQIEWzuL3HAw+inRbutod2gMxDrJpLgK+qUW4ydxkpNZoPbDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9f2384634ea873e603562deb18cd14514701c6732872a7afb3d6cd86ddd7481","last_reissued_at":"2026-07-05T08:23:23.975560Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:23.975560Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Worldwide Federated Training of Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Alex Iacob, Bill Marino, Lorenzo Sani, Nicholas Donald Lane, Preslav Aleksandrov, William F. Shen","submitted_at":"2024-05-23T11:25:19Z","abstract_excerpt":"The reliance of language model training on massive amounts of computation and vast datasets scraped from potentially low-quality, copyrighted, or sensitive data has come into question practically, legally, and ethically. Federated learning provides a plausible alternative by enabling previously untapped data to be voluntarily gathered from collaborating organizations. However, when scaled globally, federated learning requires collaboration across heterogeneous legal, security, and privacy regimes while accounting for the inherent locality of language data; this further exacerbates the establis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14446","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14446","created_at":"2026-07-05T08:23:23.975631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14446v2","created_at":"2026-07-05T08:23:23.975631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14446","created_at":"2026-07-05T08:23:23.975631+00:00"},{"alias_kind":"pith_short_12","alias_value":"XHZDQRRU5KDT","created_at":"2026-07-05T08:23:23.975631+00:00"},{"alias_kind":"pith_short_16","alias_value":"XHZDQRRU5KDT4YBV","created_at":"2026-07-05T08:23:23.975631+00:00"},{"alias_kind":"pith_short_8","alias_value":"XHZDQRRU","created_at":"2026-07-05T08:23:23.975631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.00050","citing_title":"Task-Centric Personalized Federated Fine-Tuning of Language Models","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU","json":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU.json","graph_json":"https://pith.science/api/pith-number/XHZDQRRU5KDT4YBVMLPLDDGRIU/graph.json","events_json":"https://pith.science/api/pith-number/XHZDQRRU5KDT4YBVMLPLDDGRIU/events.json","paper":"https://pith.science/paper/XHZDQRRU"},"agent_actions":{"view_html":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU","download_json":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU.json","view_paper":"https://pith.science/paper/XHZDQRRU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14446&json=true","fetch_graph":"https://pith.science/api/pith-number/XHZDQRRU5KDT4YBVMLPLDDGRIU/graph.json","fetch_events":"https://pith.science/api/pith-number/XHZDQRRU5KDT4YBVMLPLDDGRIU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU/action/storage_attestation","attest_author":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU/action/author_attestation","sign_citation":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU/action/citation_signature","submit_replication":"https://pith.science/pith/XHZDQRRU5KDT4YBVMLPLDDGRIU/action/replication_record"}},"created_at":"2026-07-05T08:23:23.975631+00:00","updated_at":"2026-07-05T08:23:23.975631+00:00"}