{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:B2U56Z5ZGO4RTXN3A4MX65Q7CJ","short_pith_number":"pith:B2U56Z5Z","schema_version":"1.0","canonical_sha256":"0ea9df67b933b919ddbb07197f761f125aff1bba3375b5c7338fb310e2f89690","source":{"kind":"arxiv","id":"2312.15023","version":2},"attestation_state":"computed","paper":{"title":"Federated Q-Learning: Linear Regret Speedup with Low Communication Cost","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Fengyu Gao, Jing Yang, Lingzhou Xue, Zhong Zheng","submitted_at":"2023-12-22T19:14:09Z","abstract_excerpt":"In this paper, we consider federated reinforcement learning for tabular episodic Markov Decision Processes (MDP) where, under the coordination of a central server, multiple agents collaboratively explore the environment and learn an optimal policy without sharing their raw data. While linear speedup in the number of agents has been achieved for some metrics, such as convergence rate and sample complexity, in similar settings, it is unclear whether it is possible to design a model-free algorithm to achieve linear regret speedup with low communication cost. We propose two federated Q-Learning al"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15023","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-22T19:14:09Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"8ec55d2d28a4cb60d505732098edd97cf56506c1cf2332eed5432a6e35dd3b9a","abstract_canon_sha256":"6cf750ecc3894376191162385f0546543dde3fa6d2c0dc6ecafe5cf74f6f6252"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:16:45.703030Z","signature_b64":"H2V7k9/GtV7+rOtYN3rgPhJNzlVDygKZqRXmQ1eZIvntTexPdU9Jh1Zr3Hkg3seQgvb3hGc+Qk71WdBzKyypBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ea9df67b933b919ddbb07197f761f125aff1bba3375b5c7338fb310e2f89690","last_reissued_at":"2026-07-05T08:16:45.702558Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:16:45.702558Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Federated Q-Learning: Linear Regret Speedup with Low Communication Cost","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Fengyu Gao, Jing Yang, Lingzhou Xue, Zhong Zheng","submitted_at":"2023-12-22T19:14:09Z","abstract_excerpt":"In this paper, we consider federated reinforcement learning for tabular episodic Markov Decision Processes (MDP) where, under the coordination of a central server, multiple agents collaboratively explore the environment and learn an optimal policy without sharing their raw data. While linear speedup in the number of agents has been achieved for some metrics, such as convergence rate and sample complexity, in similar settings, it is unclear whether it is possible to design a model-free algorithm to achieve linear regret speedup with low communication cost. We propose two federated Q-Learning al"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15023","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15023/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15023","created_at":"2026-07-05T08:16:45.702612+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15023v2","created_at":"2026-07-05T08:16:45.702612+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15023","created_at":"2026-07-05T08:16:45.702612+00:00"},{"alias_kind":"pith_short_12","alias_value":"B2U56Z5ZGO4R","created_at":"2026-07-05T08:16:45.702612+00:00"},{"alias_kind":"pith_short_16","alias_value":"B2U56Z5ZGO4RTXN3","created_at":"2026-07-05T08:16:45.702612+00:00"},{"alias_kind":"pith_short_8","alias_value":"B2U56Z5Z","created_at":"2026-07-05T08:16:45.702612+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.03897","citing_title":"On the Convergence Rates of Federated Q-Learning across Heterogeneous Environments","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ","json":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ.json","graph_json":"https://pith.science/api/pith-number/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/graph.json","events_json":"https://pith.science/api/pith-number/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/events.json","paper":"https://pith.science/paper/B2U56Z5Z"},"agent_actions":{"view_html":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ","download_json":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ.json","view_paper":"https://pith.science/paper/B2U56Z5Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15023&json=true","fetch_graph":"https://pith.science/api/pith-number/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/graph.json","fetch_events":"https://pith.science/api/pith-number/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/action/storage_attestation","attest_author":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/action/author_attestation","sign_citation":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/action/citation_signature","submit_replication":"https://pith.science/pith/B2U56Z5ZGO4RTXN3A4MX65Q7CJ/action/replication_record"}},"created_at":"2026-07-05T08:16:45.702612+00:00","updated_at":"2026-07-05T08:16:45.702612+00:00"}