{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:U672K43KOIAYMTR6LNRU7RPVTM","short_pith_number":"pith:U672K43K","canonical_record":{"source":{"id":"2405.00282","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-05-01T02:19:31Z","cross_cats_sorted":["cs.AI","cs.GT","cs.LG","cs.MA"],"title_canon_sha256":"34c1c2fcf3da8cbd83b14dd84a13c1f4be1fb3d648f1298598a5b0fd58067c22","abstract_canon_sha256":"3d54c8f8c0104c1093dd14f9fb476715d9e6feaa23bfc105fd0ad950f5c182d0"},"schema_version":"1.0"},"canonical_sha256":"a7bfa5736a7201864e3e5b634fc5f59b34b2b26c40852d075fa469ca95fbd546","source":{"kind":"arxiv","id":"2405.00282","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.00282","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"arxiv_version","alias_value":"2405.00282v2","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00282","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"pith_short_12","alias_value":"U672K43KOIAY","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"pith_short_16","alias_value":"U672K43KOIAYMTR6","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"pith_short_8","alias_value":"U672K43K","created_at":"2026-07-05T12:03:42Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:U672K43KOIAYMTR6LNRU7RPVTM","target":"record","payload":{"canonical_record":{"source":{"id":"2405.00282","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-05-01T02:19:31Z","cross_cats_sorted":["cs.AI","cs.GT","cs.LG","cs.MA"],"title_canon_sha256":"34c1c2fcf3da8cbd83b14dd84a13c1f4be1fb3d648f1298598a5b0fd58067c22","abstract_canon_sha256":"3d54c8f8c0104c1093dd14f9fb476715d9e6feaa23bfc105fd0ad950f5c182d0"},"schema_version":"1.0"},"canonical_sha256":"a7bfa5736a7201864e3e5b634fc5f59b34b2b26c40852d075fa469ca95fbd546","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:42.463239Z","signature_b64":"ljN+FLzgAqFgYNT/ulf1YiZe2NDQCHXhDYRHptOrIZbhTOFZbVAXp0Uptpr+9otvbPri8Eo3v+WrG1bcKW61BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7bfa5736a7201864e3e5b634fc5f59b34b2b26c40852d075fa469ca95fbd546","last_reissued_at":"2026-07-05T12:03:42.462770Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:42.462770Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.00282","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:03:42Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"oE1CsIEXd6uMNEmEkhXrK/EwdIscvRoGcuJMgnAJKxCASEZvocEtJbhz/uDt1AMnV//9s/Q8ZergcKohFzhjDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T16:10:02.220979Z"},"content_sha256":"165015e5b7713faf296e1e28f62d0c886c9cf1d3b7a2df7ef83b5984f8bf8add","schema_version":"1.0","event_id":"sha256:165015e5b7713faf296e1e28f62d0c886c9cf1d3b7a2df7ef83b5984f8bf8add"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:U672K43KOIAYMTR6LNRU7RPVTM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"MF-OML: Online Mean-Field Reinforcement Learning with Occupation Measures for Large Population Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT","cs.LG","cs.MA"],"primary_cat":"math.OC","authors_text":"Anran Hu, Junzi Zhang","submitted_at":"2024-05-01T02:19:31Z","abstract_excerpt":"Reinforcement learning for multi-agent games has attracted lots of attention recently. However, given the challenge of solving Nash equilibria for large population games, existing works with guaranteed polynomial complexities either focus on variants of zero-sum and potential games, or aim at solving (coarse) correlated equilibria, or require access to simulators, or rely on certain assumptions that are hard to verify. This work proposes MF-OML (Mean-Field Occupation-Measure Learning), an online mean-field reinforcement learning algorithm for computing approximate Nash equilibria of large popu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00282","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.00282/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:03:42Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Xx1kRJgksFbe5m2y6rler//LIYmx5NwOm3A8O9Gb/gW1MYZHj/UFP9JvZMr3+Rm11hFgGAXrozHV1DxGt+BfCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T16:10:02.221554Z"},"content_sha256":"83d64d60d7bf8e317d245322db5b648981af508c7808be9441010d12e9d03d59","schema_version":"1.0","event_id":"sha256:83d64d60d7bf8e317d245322db5b648981af508c7808be9441010d12e9d03d59"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/U672K43KOIAYMTR6LNRU7RPVTM/bundle.json","state_url":"https://pith.science/pith/U672K43KOIAYMTR6LNRU7RPVTM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/U672K43KOIAYMTR6LNRU7RPVTM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T16:10:02Z","links":{"resolver":"https://pith.science/pith/U672K43KOIAYMTR6LNRU7RPVTM","bundle":"https://pith.science/pith/U672K43KOIAYMTR6LNRU7RPVTM/bundle.json","state":"https://pith.science/pith/U672K43KOIAYMTR6LNRU7RPVTM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/U672K43KOIAYMTR6LNRU7RPVTM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:U672K43KOIAYMTR6LNRU7RPVTM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3d54c8f8c0104c1093dd14f9fb476715d9e6feaa23bfc105fd0ad950f5c182d0","cross_cats_sorted":["cs.AI","cs.GT","cs.LG","cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-05-01T02:19:31Z","title_canon_sha256":"34c1c2fcf3da8cbd83b14dd84a13c1f4be1fb3d648f1298598a5b0fd58067c22"},"schema_version":"1.0","source":{"id":"2405.00282","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.00282","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"arxiv_version","alias_value":"2405.00282v2","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00282","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"pith_short_12","alias_value":"U672K43KOIAY","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"pith_short_16","alias_value":"U672K43KOIAYMTR6","created_at":"2026-07-05T12:03:42Z"},{"alias_kind":"pith_short_8","alias_value":"U672K43K","created_at":"2026-07-05T12:03:42Z"}],"graph_snapshots":[{"event_id":"sha256:83d64d60d7bf8e317d245322db5b648981af508c7808be9441010d12e9d03d59","target":"graph","created_at":"2026-07-05T12:03:42Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.00282/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning for multi-agent games has attracted lots of attention recently. However, given the challenge of solving Nash equilibria for large population games, existing works with guaranteed polynomial complexities either focus on variants of zero-sum and potential games, or aim at solving (coarse) correlated equilibria, or require access to simulators, or rely on certain assumptions that are hard to verify. This work proposes MF-OML (Mean-Field Occupation-Measure Learning), an online mean-field reinforcement learning algorithm for computing approximate Nash equilibria of large popu","authors_text":"Anran Hu, Junzi Zhang","cross_cats":["cs.AI","cs.GT","cs.LG","cs.MA"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-05-01T02:19:31Z","title":"MF-OML: Online Mean-Field Reinforcement Learning with Occupation Measures for Large Population Games"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00282","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:165015e5b7713faf296e1e28f62d0c886c9cf1d3b7a2df7ef83b5984f8bf8add","target":"record","created_at":"2026-07-05T12:03:42Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3d54c8f8c0104c1093dd14f9fb476715d9e6feaa23bfc105fd0ad950f5c182d0","cross_cats_sorted":["cs.AI","cs.GT","cs.LG","cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-05-01T02:19:31Z","title_canon_sha256":"34c1c2fcf3da8cbd83b14dd84a13c1f4be1fb3d648f1298598a5b0fd58067c22"},"schema_version":"1.0","source":{"id":"2405.00282","kind":"arxiv","version":2}},"canonical_sha256":"a7bfa5736a7201864e3e5b634fc5f59b34b2b26c40852d075fa469ca95fbd546","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a7bfa5736a7201864e3e5b634fc5f59b34b2b26c40852d075fa469ca95fbd546","first_computed_at":"2026-07-05T12:03:42.462770Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:03:42.462770Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ljN+FLzgAqFgYNT/ulf1YiZe2NDQCHXhDYRHptOrIZbhTOFZbVAXp0Uptpr+9otvbPri8Eo3v+WrG1bcKW61BA==","signature_status":"signed_v1","signed_at":"2026-07-05T12:03:42.463239Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.00282","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:165015e5b7713faf296e1e28f62d0c886c9cf1d3b7a2df7ef83b5984f8bf8add","sha256:83d64d60d7bf8e317d245322db5b648981af508c7808be9441010d12e9d03d59"],"state_sha256":"fcddc7551cf80738968d86f566e7547bf709f14f50ff17c5b41a27f44a6700b2"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Gr0w2ygAavgIUQEDDcokAasyWz4pkxJjHWC7oxvoPoLPm3QL8hcyrjzNVepldvxo8Fg6MRtVTwo3TGLgHe1mBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T16:10:02.225086Z","bundle_sha256":"8968ed67e30db91e243f74aa4fceffeecf9faa075af8b9519b5f05ed67d38bfb"}}