{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5SY7IKX2KWQ4C7KFBXXP4EB76A","short_pith_number":"pith:5SY7IKX2","schema_version":"1.0","canonical_sha256":"ecb1f42afa55a1c17d450deefe103ff03a40498f65b8d4e014473db595f69041","source":{"kind":"arxiv","id":"2403.08100","version":1},"attestation_state":"computed","paper":{"title":"Efficient Language Model Architectures for Differentially Private Federated Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.DC"],"primary_cat":"cs.LG","authors_text":"Ananda Theertha Suresh, Jae Hun Ro, Srinadh Bhojanapalli, Yanxiang Zhang, Zheng Xu","submitted_at":"2024-03-12T22:21:48Z","abstract_excerpt":"Cross-device federated learning (FL) is a technique that trains a model on data distributed across typically millions of edge devices without data leaving the devices. SGD is the standard client optimizer for on device training in cross-device FL, favored for its memory and computational efficiency. However, in centralized training of neural language models, adaptive optimizers are preferred as they offer improved stability and performance. In light of this, we ask if language models can be modified such that they can be efficiently trained with SGD client optimizers and answer this affirmativ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08100","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-12T22:21:48Z","cross_cats_sorted":["cs.CR","cs.DC"],"title_canon_sha256":"de577616a178110a5e4246634a783851c2bce7e8c12cbd260f851c1292c2c527","abstract_canon_sha256":"17fed4f6beea7af271b55a62ef467cc34b14af7afd428518329562e1aafb2375"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:55:34.340242Z","signature_b64":"h51zyWLsm8UXcVmvansdsg6tbikQGkbNw8dtV4Boc/sFNCWeTi0O5l3eQjED/bOLXIkVSsiLZdYaRjPSq5jMBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecb1f42afa55a1c17d450deefe103ff03a40498f65b8d4e014473db595f69041","last_reissued_at":"2026-07-05T07:55:34.339892Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:55:34.339892Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Language Model Architectures for Differentially Private Federated Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.DC"],"primary_cat":"cs.LG","authors_text":"Ananda Theertha Suresh, Jae Hun Ro, Srinadh Bhojanapalli, Yanxiang Zhang, Zheng Xu","submitted_at":"2024-03-12T22:21:48Z","abstract_excerpt":"Cross-device federated learning (FL) is a technique that trains a model on data distributed across typically millions of edge devices without data leaving the devices. SGD is the standard client optimizer for on device training in cross-device FL, favored for its memory and computational efficiency. However, in centralized training of neural language models, adaptive optimizers are preferred as they offer improved stability and performance. In light of this, we ask if language models can be modified such that they can be efficiently trained with SGD client optimizers and answer this affirmativ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08100","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08100","created_at":"2026-07-05T07:55:34.339949+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08100v1","created_at":"2026-07-05T07:55:34.339949+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08100","created_at":"2026-07-05T07:55:34.339949+00:00"},{"alias_kind":"pith_short_12","alias_value":"5SY7IKX2KWQ4","created_at":"2026-07-05T07:55:34.339949+00:00"},{"alias_kind":"pith_short_16","alias_value":"5SY7IKX2KWQ4C7KF","created_at":"2026-07-05T07:55:34.339949+00:00"},{"alias_kind":"pith_short_8","alias_value":"5SY7IKX2","created_at":"2026-07-05T07:55:34.339949+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.12699","citing_title":"SoK: The Privacy Paradox of Large Language Models: Advancements, Privacy Risks, and Mitigation","ref_index":95,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A","json":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A.json","graph_json":"https://pith.science/api/pith-number/5SY7IKX2KWQ4C7KFBXXP4EB76A/graph.json","events_json":"https://pith.science/api/pith-number/5SY7IKX2KWQ4C7KFBXXP4EB76A/events.json","paper":"https://pith.science/paper/5SY7IKX2"},"agent_actions":{"view_html":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A","download_json":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A.json","view_paper":"https://pith.science/paper/5SY7IKX2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08100&json=true","fetch_graph":"https://pith.science/api/pith-number/5SY7IKX2KWQ4C7KFBXXP4EB76A/graph.json","fetch_events":"https://pith.science/api/pith-number/5SY7IKX2KWQ4C7KFBXXP4EB76A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A/action/storage_attestation","attest_author":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A/action/author_attestation","sign_citation":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A/action/citation_signature","submit_replication":"https://pith.science/pith/5SY7IKX2KWQ4C7KFBXXP4EB76A/action/replication_record"}},"created_at":"2026-07-05T07:55:34.339949+00:00","updated_at":"2026-07-05T07:55:34.339949+00:00"}