{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:7IQLZSEG72MFRS6IHIKEO5L6TQ","short_pith_number":"pith:7IQLZSEG","schema_version":"1.0","canonical_sha256":"fa20bcc886fe9858cbc83a1447757e9c38e3dc87c45ce9aad0125936b4e2d73f","source":{"kind":"arxiv","id":"2107.04050","version":2},"attestation_state":"computed","paper":{"title":"Efficient Model-Based Multi-Agent Mean-Field Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"stat.ML","authors_text":"Andreas Krause, Barna P\\'asztor, Ilija Bogunovic","submitted_at":"2021-07-08T18:01:02Z","abstract_excerpt":"Learning in multi-agent systems is highly challenging due to several factors including the non-stationarity introduced by agents' interactions and the combinatorial nature of their state and action spaces. In particular, we consider the Mean-Field Control (MFC) problem which assumes an asymptotically infinite population of identical agents that aim to collaboratively maximize the collective reward. In many cases, solutions of an MFC problem are good approximations for large systems, hence, efficient learning for MFC is valuable for the analogous discrete agent setting with many agents. Specifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.04050","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2021-07-08T18:01:02Z","cross_cats_sorted":["cs.LG","cs.MA"],"title_canon_sha256":"0f9b4ea43bd1b9f0f49e46d5c3d14a261ce1e8b0a234b1b3533cf6a8b71d856b","abstract_canon_sha256":"e503e6ab3c44faebc02a4f41e83104fc6a2f2280c071b993e4679fc5e96c0524"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:08:02.104068Z","signature_b64":"Bg/0tysToqYPgfDRsjIphVy1E9wHvdgTYfgchjeH6Y/tMWb9lnLPMYdXKbicIKg4zo09XvjLjQxaYw/FcgA1Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa20bcc886fe9858cbc83a1447757e9c38e3dc87c45ce9aad0125936b4e2d73f","last_reissued_at":"2026-07-05T06:08:02.103671Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:08:02.103671Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Model-Based Multi-Agent Mean-Field Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"stat.ML","authors_text":"Andreas Krause, Barna P\\'asztor, Ilija Bogunovic","submitted_at":"2021-07-08T18:01:02Z","abstract_excerpt":"Learning in multi-agent systems is highly challenging due to several factors including the non-stationarity introduced by agents' interactions and the combinatorial nature of their state and action spaces. In particular, we consider the Mean-Field Control (MFC) problem which assumes an asymptotically infinite population of identical agents that aim to collaboratively maximize the collective reward. In many cases, solutions of an MFC problem are good approximations for large systems, hence, efficient learning for MFC is valuable for the analogous discrete agent setting with many agents. Specifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.04050","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.04050/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.04050","created_at":"2026-07-05T06:08:02.103738+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.04050v2","created_at":"2026-07-05T06:08:02.103738+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.04050","created_at":"2026-07-05T06:08:02.103738+00:00"},{"alias_kind":"pith_short_12","alias_value":"7IQLZSEG72MF","created_at":"2026-07-05T06:08:02.103738+00:00"},{"alias_kind":"pith_short_16","alias_value":"7IQLZSEG72MFRS6I","created_at":"2026-07-05T06:08:02.103738+00:00"},{"alias_kind":"pith_short_8","alias_value":"7IQLZSEG","created_at":"2026-07-05T06:08:02.103738+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.15103","citing_title":"Vulnerable Agent Identification in Large-Scale Multi-Agent Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27378","citing_title":"Continuous-time q-learning for mean-field control with common noise, part-II: q-learning algorithms","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ","json":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ.json","graph_json":"https://pith.science/api/pith-number/7IQLZSEG72MFRS6IHIKEO5L6TQ/graph.json","events_json":"https://pith.science/api/pith-number/7IQLZSEG72MFRS6IHIKEO5L6TQ/events.json","paper":"https://pith.science/paper/7IQLZSEG"},"agent_actions":{"view_html":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ","download_json":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ.json","view_paper":"https://pith.science/paper/7IQLZSEG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.04050&json=true","fetch_graph":"https://pith.science/api/pith-number/7IQLZSEG72MFRS6IHIKEO5L6TQ/graph.json","fetch_events":"https://pith.science/api/pith-number/7IQLZSEG72MFRS6IHIKEO5L6TQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ/action/storage_attestation","attest_author":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ/action/author_attestation","sign_citation":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ/action/citation_signature","submit_replication":"https://pith.science/pith/7IQLZSEG72MFRS6IHIKEO5L6TQ/action/replication_record"}},"created_at":"2026-07-05T06:08:02.103738+00:00","updated_at":"2026-07-05T06:08:02.103738+00:00"}