{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YVSZ5RA5MHZQPYOYMCUY7I6WC4","short_pith_number":"pith:YVSZ5RA5","schema_version":"1.0","canonical_sha256":"c5659ec41d61f307e1d860a98fa3d61722a8274322da0420d38a7ba49830f0cf","source":{"kind":"arxiv","id":"2312.06659","version":2},"attestation_state":"computed","paper":{"title":"Convergence of Multi-Scale Reinforcement Q-Learning Algorithms for Mean Field Game and Control Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Andrea Angiuli, Jean-Pierre Fouque, Mathieu Lauri\\`ere, Mengrui Zhang","submitted_at":"2023-12-11T18:59:51Z","abstract_excerpt":"We establish the convergence of the unified two-timescale Reinforcement Learning (RL) algorithm presented in a previous work by Angiuli et al. This algorithm provides solutions to Mean Field Game (MFG) or Mean Field Control (MFC) problems depending on the ratio of two learning rates, one for the value function and the other for the mean field term. Our proof of convergence highlights the fact that in the case of MFC several mean field distributions need to be updated and for this reason we present two separate algorithms, one for MFG and one for MFC. We focus on a setting with finite state and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.06659","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2023-12-11T18:59:51Z","cross_cats_sorted":[],"title_canon_sha256":"f9f380dba321bf6ac6d0d47e04b968576fe3976deb5e03539330819c8ba4708e","abstract_canon_sha256":"404b47ce52d72ec7bea594be907492e25282819b911a322965f0c44ee8eab392"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:14:02.701488Z","signature_b64":"J/pOp1OVmi0w6/BCEi5HWolVuRWbZ6y12H54uIT1Ks/OhzaRP+fNqDrgjOVDOdLQVgoYlWkIm5F4ijybDMB5Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5659ec41d61f307e1d860a98fa3d61722a8274322da0420d38a7ba49830f0cf","last_reissued_at":"2026-07-05T08:14:02.700984Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:14:02.700984Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Convergence of Multi-Scale Reinforcement Q-Learning Algorithms for Mean Field Game and Control Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Andrea Angiuli, Jean-Pierre Fouque, Mathieu Lauri\\`ere, Mengrui Zhang","submitted_at":"2023-12-11T18:59:51Z","abstract_excerpt":"We establish the convergence of the unified two-timescale Reinforcement Learning (RL) algorithm presented in a previous work by Angiuli et al. This algorithm provides solutions to Mean Field Game (MFG) or Mean Field Control (MFC) problems depending on the ratio of two learning rates, one for the value function and the other for the mean field term. Our proof of convergence highlights the fact that in the case of MFC several mean field distributions need to be updated and for this reason we present two separate algorithms, one for MFG and one for MFC. We focus on a setting with finite state and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.06659","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.06659/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.06659","created_at":"2026-07-05T08:14:02.701049+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.06659v2","created_at":"2026-07-05T08:14:02.701049+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.06659","created_at":"2026-07-05T08:14:02.701049+00:00"},{"alias_kind":"pith_short_12","alias_value":"YVSZ5RA5MHZQ","created_at":"2026-07-05T08:14:02.701049+00:00"},{"alias_kind":"pith_short_16","alias_value":"YVSZ5RA5MHZQPYOY","created_at":"2026-07-05T08:14:02.701049+00:00"},{"alias_kind":"pith_short_8","alias_value":"YVSZ5RA5","created_at":"2026-07-05T08:14:02.701049+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01525","citing_title":"Mean Field Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27378","citing_title":"Continuous-time q-learning for mean-field control with common noise, part-II: q-learning algorithms","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4","json":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4.json","graph_json":"https://pith.science/api/pith-number/YVSZ5RA5MHZQPYOYMCUY7I6WC4/graph.json","events_json":"https://pith.science/api/pith-number/YVSZ5RA5MHZQPYOYMCUY7I6WC4/events.json","paper":"https://pith.science/paper/YVSZ5RA5"},"agent_actions":{"view_html":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4","download_json":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4.json","view_paper":"https://pith.science/paper/YVSZ5RA5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.06659&json=true","fetch_graph":"https://pith.science/api/pith-number/YVSZ5RA5MHZQPYOYMCUY7I6WC4/graph.json","fetch_events":"https://pith.science/api/pith-number/YVSZ5RA5MHZQPYOYMCUY7I6WC4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4/action/storage_attestation","attest_author":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4/action/author_attestation","sign_citation":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4/action/citation_signature","submit_replication":"https://pith.science/pith/YVSZ5RA5MHZQPYOYMCUY7I6WC4/action/replication_record"}},"created_at":"2026-07-05T08:14:02.701049+00:00","updated_at":"2026-07-05T08:14:02.701049+00:00"}