{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:ZTM4EVFXJFNXNNDOF2SUEJB54F","short_pith_number":"pith:ZTM4EVFX","canonical_record":{"source":{"id":"2408.02489","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-08-05T14:11:51Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"5f2932043f022097b624e7c3c90b14cb92e99b361fd86b110eeea2056190e213","abstract_canon_sha256":"8130c6f912264bc134db1555c3e5906c2d46f22ae2506d6a53975ac9e7bc81ce"},"schema_version":"1.0"},"canonical_sha256":"ccd9c254b7495b76b46e2ea542243de1538af45d6000cfddc3d6bcc8447574ed","source":{"kind":"arxiv","id":"2408.02489","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2408.02489","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"arxiv_version","alias_value":"2408.02489v1","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.02489","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"pith_short_12","alias_value":"ZTM4EVFXJFNX","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"pith_short_16","alias_value":"ZTM4EVFXJFNXNNDO","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"pith_short_8","alias_value":"ZTM4EVFX","created_at":"2026-07-05T08:52:13Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:ZTM4EVFXJFNXNNDOF2SUEJB54F","target":"record","payload":{"canonical_record":{"source":{"id":"2408.02489","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-08-05T14:11:51Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"5f2932043f022097b624e7c3c90b14cb92e99b361fd86b110eeea2056190e213","abstract_canon_sha256":"8130c6f912264bc134db1555c3e5906c2d46f22ae2506d6a53975ac9e7bc81ce"},"schema_version":"1.0"},"canonical_sha256":"ccd9c254b7495b76b46e2ea542243de1538af45d6000cfddc3d6bcc8447574ed","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:52:13.478614Z","signature_b64":"5R4XrnLMk1ipmjYwSpS/wV/K6K555L3Yo6JkhkU5v5XlTSbMlx+aogOPc5yhhw14dlMpkfbEWNqTdSiH/8a4AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ccd9c254b7495b76b46e2ea542243de1538af45d6000cfddc3d6bcc8447574ed","last_reissued_at":"2026-07-05T08:52:13.478079Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:52:13.478079Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2408.02489","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:52:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JJD/yntcpJvSQfZirYPOb9dBJ7nrdnH0ZXzLTKnQSnja/uqA9rBu1GLKXRiyK2DSYyImG8EBY6zwlt7BC1XFDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T08:10:58.868316Z"},"content_sha256":"a17d9884943753809f9dac7d1b7c56fb710bce0c6f2dcd1a8a2160ddd8a39b09","schema_version":"1.0","event_id":"sha256:a17d9884943753809f9dac7d1b7c56fb710bce0c6f2dcd1a8a2160ddd8a39b09"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:ZTM4EVFXJFNXNNDOF2SUEJB54F","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Full error analysis of policy gradient learning algorithms for exploratory linear quadratic mean-field control problem in continuous time with common noise","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"math.OC","authors_text":"Huy\\^en Pham (LPSM (UMR\\_8001)), Noufel Frikha (CES), Xuanye Song (LPSM (UMR\\_8001))","submitted_at":"2024-08-05T14:11:51Z","abstract_excerpt":"We consider reinforcement learning (RL) methods for finding optimal policies in linear quadratic (LQ) mean field control (MFC) problems over an infinite horizon in continuous time, with common noise and entropy regularization. We study policy gradient (PG) learning and first demonstrate convergence in a model-based setting by establishing a suitable gradient domination condition.Next, our main contribution is a comprehensive error analysis, where we prove the global linear convergence and sample complexity of the PG algorithm with two-point gradient estimates in a model-free setting with unkno"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.02489","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.02489/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:52:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aFHh5IaY4ChLwcEfPEBQ81jdJCwEUpyf5r9qR3pbGjGsgw46BTvdW8yg44ETVFnBtfUxZAdNTCnweWprEZ5SAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T08:10:58.868873Z"},"content_sha256":"03cec273649163017e8dc6d8083ddaf193577f70c3c9ac6f8d8031f9bf88b2ee","schema_version":"1.0","event_id":"sha256:03cec273649163017e8dc6d8083ddaf193577f70c3c9ac6f8d8031f9bf88b2ee"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ZTM4EVFXJFNXNNDOF2SUEJB54F/bundle.json","state_url":"https://pith.science/pith/ZTM4EVFXJFNXNNDOF2SUEJB54F/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ZTM4EVFXJFNXNNDOF2SUEJB54F/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T08:10:58Z","links":{"resolver":"https://pith.science/pith/ZTM4EVFXJFNXNNDOF2SUEJB54F","bundle":"https://pith.science/pith/ZTM4EVFXJFNXNNDOF2SUEJB54F/bundle.json","state":"https://pith.science/pith/ZTM4EVFXJFNXNNDOF2SUEJB54F/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ZTM4EVFXJFNXNNDOF2SUEJB54F/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:ZTM4EVFXJFNXNNDOF2SUEJB54F","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8130c6f912264bc134db1555c3e5906c2d46f22ae2506d6a53975ac9e7bc81ce","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-08-05T14:11:51Z","title_canon_sha256":"5f2932043f022097b624e7c3c90b14cb92e99b361fd86b110eeea2056190e213"},"schema_version":"1.0","source":{"id":"2408.02489","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2408.02489","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"arxiv_version","alias_value":"2408.02489v1","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.02489","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"pith_short_12","alias_value":"ZTM4EVFXJFNX","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"pith_short_16","alias_value":"ZTM4EVFXJFNXNNDO","created_at":"2026-07-05T08:52:13Z"},{"alias_kind":"pith_short_8","alias_value":"ZTM4EVFX","created_at":"2026-07-05T08:52:13Z"}],"graph_snapshots":[{"event_id":"sha256:03cec273649163017e8dc6d8083ddaf193577f70c3c9ac6f8d8031f9bf88b2ee","target":"graph","created_at":"2026-07-05T08:52:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2408.02489/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We consider reinforcement learning (RL) methods for finding optimal policies in linear quadratic (LQ) mean field control (MFC) problems over an infinite horizon in continuous time, with common noise and entropy regularization. We study policy gradient (PG) learning and first demonstrate convergence in a model-based setting by establishing a suitable gradient domination condition.Next, our main contribution is a comprehensive error analysis, where we prove the global linear convergence and sample complexity of the PG algorithm with two-point gradient estimates in a model-free setting with unkno","authors_text":"Huy\\^en Pham (LPSM (UMR\\_8001)), Noufel Frikha (CES), Xuanye Song (LPSM (UMR\\_8001))","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-08-05T14:11:51Z","title":"Full error analysis of policy gradient learning algorithms for exploratory linear quadratic mean-field control problem in continuous time with common noise"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.02489","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a17d9884943753809f9dac7d1b7c56fb710bce0c6f2dcd1a8a2160ddd8a39b09","target":"record","created_at":"2026-07-05T08:52:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8130c6f912264bc134db1555c3e5906c2d46f22ae2506d6a53975ac9e7bc81ce","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-08-05T14:11:51Z","title_canon_sha256":"5f2932043f022097b624e7c3c90b14cb92e99b361fd86b110eeea2056190e213"},"schema_version":"1.0","source":{"id":"2408.02489","kind":"arxiv","version":1}},"canonical_sha256":"ccd9c254b7495b76b46e2ea542243de1538af45d6000cfddc3d6bcc8447574ed","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ccd9c254b7495b76b46e2ea542243de1538af45d6000cfddc3d6bcc8447574ed","first_computed_at":"2026-07-05T08:52:13.478079Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:52:13.478079Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"5R4XrnLMk1ipmjYwSpS/wV/K6K555L3Yo6JkhkU5v5XlTSbMlx+aogOPc5yhhw14dlMpkfbEWNqTdSiH/8a4AQ==","signature_status":"signed_v1","signed_at":"2026-07-05T08:52:13.478614Z","signed_message":"canonical_sha256_bytes"},"source_id":"2408.02489","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a17d9884943753809f9dac7d1b7c56fb710bce0c6f2dcd1a8a2160ddd8a39b09","sha256:03cec273649163017e8dc6d8083ddaf193577f70c3c9ac6f8d8031f9bf88b2ee"],"state_sha256":"a655fe83b3f87bf96cdbaa9cddf383a8f9256c5ceaa40b43e6abf927b93b38c2"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"DECLSAQknMpgaao73QFOWcf5jnkb0an2uZ2WlW/h75o8o+2uRcFTLH6RHiGRS+6COKELpkeDr4Yh06EsIruuAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T08:10:58.873789Z","bundle_sha256":"cfc0d4c52c739bb5e7bec7bfcdd7ae805326ea4247b5f473abad374fd73af9b2"}}