{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:UX2V6BZ2RTN3B3QVEXJNCRJK3K","short_pith_number":"pith:UX2V6BZ2","schema_version":"1.0","canonical_sha256":"a5f55f073a8cdbb0ee1525d2d1452adaa415a3de570ba0a8b1cd403269d2d6f0","source":{"kind":"arxiv","id":"2006.13912","version":3},"attestation_state":"computed","paper":{"title":"Unified Reinforcement Q-Learning for Mean Field Game and Control Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"math.OC","authors_text":"Andrea Angiuli, Jean-Pierre Fouque, Mathieu Lauri\\`ere","submitted_at":"2020-06-24T17:45:44Z","abstract_excerpt":"We present a Reinforcement Learning (RL) algorithm to solve infinite horizon asymptotic Mean Field Game (MFG) and Mean Field Control (MFC) problems. Our approach can be described as a unified two-timescale Mean Field Q-learning: The \\emph{same} algorithm can learn either the MFG or the MFC solution by simply tuning the ratio of two learning parameters. The algorithm is in discrete time and space where the agent not only provides an action to the environment but also a distribution of the state in order to take into account the mean field feature of the problem. Importantly, we assume that the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.13912","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2020-06-24T17:45:44Z","cross_cats_sorted":["cs.LG","cs.MA"],"title_canon_sha256":"a3fe134ea2c58c2e635f127efda89937d54b48c86ce61d8cd96cbfdea122b928","abstract_canon_sha256":"bc760142b4533904868a68310b0dae1c224c6838c392b73199a7198399f50a45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:44:32.893575Z","signature_b64":"yDhf6xl9HGAN7LTW7ageECnfwA3SzjDtXdeU9ul1z4K0zQ6McrX1ka//IW5JmB3Vt03YSXrba1UFlt3fA/jrDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a5f55f073a8cdbb0ee1525d2d1452adaa415a3de570ba0a8b1cd403269d2d6f0","last_reissued_at":"2026-07-05T02:44:32.893093Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:44:32.893093Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unified Reinforcement Q-Learning for Mean Field Game and Control Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"math.OC","authors_text":"Andrea Angiuli, Jean-Pierre Fouque, Mathieu Lauri\\`ere","submitted_at":"2020-06-24T17:45:44Z","abstract_excerpt":"We present a Reinforcement Learning (RL) algorithm to solve infinite horizon asymptotic Mean Field Game (MFG) and Mean Field Control (MFC) problems. Our approach can be described as a unified two-timescale Mean Field Q-learning: The \\emph{same} algorithm can learn either the MFG or the MFC solution by simply tuning the ratio of two learning parameters. The algorithm is in discrete time and space where the agent not only provides an action to the environment but also a distribution of the state in order to take into account the mean field feature of the problem. Importantly, we assume that the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.13912","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.13912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.13912","created_at":"2026-07-05T02:44:32.893165+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.13912v3","created_at":"2026-07-05T02:44:32.893165+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.13912","created_at":"2026-07-05T02:44:32.893165+00:00"},{"alias_kind":"pith_short_12","alias_value":"UX2V6BZ2RTN3","created_at":"2026-07-05T02:44:32.893165+00:00"},{"alias_kind":"pith_short_16","alias_value":"UX2V6BZ2RTN3B3QV","created_at":"2026-07-05T02:44:32.893165+00:00"},{"alias_kind":"pith_short_8","alias_value":"UX2V6BZ2","created_at":"2026-07-05T02:44:32.893165+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.22781","citing_title":"Finite-Sample Convergence Bounds for Trust Region Policy Optimization in Mean-Field Games","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K","json":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K.json","graph_json":"https://pith.science/api/pith-number/UX2V6BZ2RTN3B3QVEXJNCRJK3K/graph.json","events_json":"https://pith.science/api/pith-number/UX2V6BZ2RTN3B3QVEXJNCRJK3K/events.json","paper":"https://pith.science/paper/UX2V6BZ2"},"agent_actions":{"view_html":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K","download_json":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K.json","view_paper":"https://pith.science/paper/UX2V6BZ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.13912&json=true","fetch_graph":"https://pith.science/api/pith-number/UX2V6BZ2RTN3B3QVEXJNCRJK3K/graph.json","fetch_events":"https://pith.science/api/pith-number/UX2V6BZ2RTN3B3QVEXJNCRJK3K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K/action/storage_attestation","attest_author":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K/action/author_attestation","sign_citation":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K/action/citation_signature","submit_replication":"https://pith.science/pith/UX2V6BZ2RTN3B3QVEXJNCRJK3K/action/replication_record"}},"created_at":"2026-07-05T02:44:32.893165+00:00","updated_at":"2026-07-05T02:44:32.893165+00:00"}