{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:UYG6U7DD2TDQCH3KZ46KC75EDT","short_pith_number":"pith:UYG6U7DD","schema_version":"1.0","canonical_sha256":"a60dea7c63d4c7011f6acf3ca17fa41cd28702f6121faad06e78bb4817115817","source":{"kind":"arxiv","id":"1910.12802","version":2},"attestation_state":"computed","paper":{"title":"Model-Free Mean-Field Reinforcement Learning: Mean-Field MDP and Mean-Field Q-Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Mathieu Lauri\\`ere, Ren\\'e Carmona, Zongjun Tan","submitted_at":"2019-10-28T16:56:46Z","abstract_excerpt":"We study infinite horizon discounted Mean Field Control (MFC) problems with common noise through the lens of Mean Field Markov Decision Processes (MFMDP). We allow the agents to use actions that are randomized not only at the individual level but also at the level of the population. This common randomization allows us to establish connections between both closed-loop and open-loop policies for MFC and Markov policies for the MFMDP. In particular, we show that there exists an optimal closed-loop policy for the original MFC. Building on this framework and the notion of state-action value functio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.12802","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2019-10-28T16:56:46Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"eca2918fb486ec5fc7be3f46f55bc183161d05826c95c159a88defbbcdc2aca4","abstract_canon_sha256":"a3f964e1ac05449973c06e5470453852ed95774e56388c55fae87a17efcdd505"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:22:18.851835Z","signature_b64":"FOGfclHa6igQ8nf7/wo8tmDDEhWvYRXU8RBNcA8MfaRkAR9jv6Xdwa4Guv7eFak3tm3Qo/q2a+W/VN1gm/GTBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a60dea7c63d4c7011f6acf3ca17fa41cd28702f6121faad06e78bb4817115817","last_reissued_at":"2026-07-05T03:22:18.850951Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:22:18.850951Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model-Free Mean-Field Reinforcement Learning: Mean-Field MDP and Mean-Field Q-Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Mathieu Lauri\\`ere, Ren\\'e Carmona, Zongjun Tan","submitted_at":"2019-10-28T16:56:46Z","abstract_excerpt":"We study infinite horizon discounted Mean Field Control (MFC) problems with common noise through the lens of Mean Field Markov Decision Processes (MFMDP). We allow the agents to use actions that are randomized not only at the individual level but also at the level of the population. This common randomization allows us to establish connections between both closed-loop and open-loop policies for MFC and Markov policies for the MFMDP. In particular, we show that there exists an optimal closed-loop policy for the original MFC. Building on this framework and the notion of state-action value functio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.12802","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.12802/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.12802","created_at":"2026-07-05T03:22:18.851469+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.12802v2","created_at":"2026-07-05T03:22:18.851469+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.12802","created_at":"2026-07-05T03:22:18.851469+00:00"},{"alias_kind":"pith_short_12","alias_value":"UYG6U7DD2TDQ","created_at":"2026-07-05T03:22:18.851469+00:00"},{"alias_kind":"pith_short_16","alias_value":"UYG6U7DD2TDQCH3K","created_at":"2026-07-05T03:22:18.851469+00:00"},{"alias_kind":"pith_short_8","alias_value":"UYG6U7DD","created_at":"2026-07-05T03:22:18.851469+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.03888","citing_title":"Measurized Markov Decision Processes","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT","json":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT.json","graph_json":"https://pith.science/api/pith-number/UYG6U7DD2TDQCH3KZ46KC75EDT/graph.json","events_json":"https://pith.science/api/pith-number/UYG6U7DD2TDQCH3KZ46KC75EDT/events.json","paper":"https://pith.science/paper/UYG6U7DD"},"agent_actions":{"view_html":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT","download_json":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT.json","view_paper":"https://pith.science/paper/UYG6U7DD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.12802&json=true","fetch_graph":"https://pith.science/api/pith-number/UYG6U7DD2TDQCH3KZ46KC75EDT/graph.json","fetch_events":"https://pith.science/api/pith-number/UYG6U7DD2TDQCH3KZ46KC75EDT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT/action/storage_attestation","attest_author":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT/action/author_attestation","sign_citation":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT/action/citation_signature","submit_replication":"https://pith.science/pith/UYG6U7DD2TDQCH3KZ46KC75EDT/action/replication_record"}},"created_at":"2026-07-05T03:22:18.851469+00:00","updated_at":"2026-07-05T03:22:18.851469+00:00"}