{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KKDGCOA3SOCINRSROFWAM2MAU7","short_pith_number":"pith:KKDGCOA3","schema_version":"1.0","canonical_sha256":"528661381b938486c651716c066980a7fa103c06b69196b0e947cef591eda454","source":{"kind":"arxiv","id":"2409.03052","version":1},"attestation_state":"computed","paper":{"title":"An Introduction to Centralized Training for Decentralized Execution in Cooperative Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Christopher Amato","submitted_at":"2024-09-04T19:54:40Z","abstract_excerpt":"Multi-agent reinforcement learning (MARL) has exploded in popularity in recent years. Many approaches have been developed but they can be divided into three main types: centralized training and execution (CTE), centralized training for decentralized execution (CTDE), and Decentralized training and execution (DTE).\n  CTDE methods are the most common as they can use centralized information during training but execute in a decentralized manner -- using only information available to that agent during execution. CTDE is the only paradigm that requires a separate training phase where any available i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.03052","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-04T19:54:40Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"fb940abca7c959eb823b377c49e980107adb6eff40145f692e3f94c59c1128f6","abstract_canon_sha256":"c9f73e151df09fba9603396d1ac72882e21833d244e6f959e8fb8a4e27728b49"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:59.327438Z","signature_b64":"M3We5u/KGRsWqc4y1B1+FUHnrsWBLZWZZbO1nvkqL1K31jwakaGkr7vWa2oXV9tWRY73k7IlFIYXXFTpImK1Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"528661381b938486c651716c066980a7fa103c06b69196b0e947cef591eda454","last_reissued_at":"2026-07-05T09:51:59.326920Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:59.326920Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Introduction to Centralized Training for Decentralized Execution in Cooperative Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Christopher Amato","submitted_at":"2024-09-04T19:54:40Z","abstract_excerpt":"Multi-agent reinforcement learning (MARL) has exploded in popularity in recent years. Many approaches have been developed but they can be divided into three main types: centralized training and execution (CTE), centralized training for decentralized execution (CTDE), and Decentralized training and execution (DTE).\n  CTDE methods are the most common as they can use centralized information during training but execute in a decentralized manner -- using only information available to that agent during execution. CTDE is the only paradigm that requires a separate training phase where any available i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.03052","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.03052/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.03052","created_at":"2026-07-05T09:51:59.326979+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.03052v1","created_at":"2026-07-05T09:51:59.326979+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.03052","created_at":"2026-07-05T09:51:59.326979+00:00"},{"alias_kind":"pith_short_12","alias_value":"KKDGCOA3SOCI","created_at":"2026-07-05T09:51:59.326979+00:00"},{"alias_kind":"pith_short_16","alias_value":"KKDGCOA3SOCINRSR","created_at":"2026-07-05T09:51:59.326979+00:00"},{"alias_kind":"pith_short_8","alias_value":"KKDGCOA3","created_at":"2026-07-05T09:51:59.326979+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07460","citing_title":"Social-spatial dependencies for learning visual navigation","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06563","citing_title":"Embodied Human-Robot Interaction via Acoustics: A MARL Approach with AcoustoBots for Spatial Data Physicalization","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18327","citing_title":"Self-CTRL: Self-Consistency Training with Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14758","citing_title":"Probabilistic Verification of Recurrent Neural Networks for Single and Multi-Agent Reinforcement Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28764","citing_title":"SwarmHarness: Skill-Based Task Routing via Decentralized Incentive-Aligned AI Agent Networks","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12352","citing_title":"CHORUS: Decentralized Multi-Embodiment Collaboration with One VLA Policy","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15400","citing_title":"Beyond Partner Diversity: An Influence-Based Team Steering Framework for Zero-Shot Human-Machine Teaming","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07548","citing_title":"Overcoming Environmental Meta-Stationarity in MARL via Adaptive Curriculum and Counterfactual Group Advantage","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2603.12031","citing_title":"AGMARL-DKS: An Adaptive Graph-Enhanced Multi-Agent Reinforcement Learning for Dynamic Kubernetes Scheduling","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13131","citing_title":"ERPPO: Entropy Regularization-based Proximal Policy Optimization","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06595","citing_title":"Cross-Modal Navigation with Multi-Agent Reinforcement Learning","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09028","citing_title":"Plasticity-Enhanced Multi-Agent Mixture of Experts for Dynamic Objective Adaptation in UAVs-Assisted Emergency Communication Networks","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17191","citing_title":"Do LLM-derived graph priors improve multi-agent coordination?","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7","json":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7.json","graph_json":"https://pith.science/api/pith-number/KKDGCOA3SOCINRSROFWAM2MAU7/graph.json","events_json":"https://pith.science/api/pith-number/KKDGCOA3SOCINRSROFWAM2MAU7/events.json","paper":"https://pith.science/paper/KKDGCOA3"},"agent_actions":{"view_html":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7","download_json":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7.json","view_paper":"https://pith.science/paper/KKDGCOA3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.03052&json=true","fetch_graph":"https://pith.science/api/pith-number/KKDGCOA3SOCINRSROFWAM2MAU7/graph.json","fetch_events":"https://pith.science/api/pith-number/KKDGCOA3SOCINRSROFWAM2MAU7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7/action/storage_attestation","attest_author":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7/action/author_attestation","sign_citation":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7/action/citation_signature","submit_replication":"https://pith.science/pith/KKDGCOA3SOCINRSROFWAM2MAU7/action/replication_record"}},"created_at":"2026-07-05T09:51:59.326979+00:00","updated_at":"2026-07-05T09:51:59.326979+00:00"}