{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:DYCFAVGNJWCX2BV3KALNTGFJ2I","short_pith_number":"pith:DYCFAVGN","schema_version":"1.0","canonical_sha256":"1e045054cd4d857d06bb5016d998a9d20a785eaf66f75f1436b3fab150b69edf","source":{"kind":"arxiv","id":"2004.11145","version":2},"attestation_state":"computed","paper":{"title":"F2A2: Flexible Fully-decentralized Approximate Actor-critic for Cooperative Multi-agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Bo Jin, Hongyuan Zha, Junchi Yan, Wenhao Li, Xiangfeng Wang","submitted_at":"2020-04-17T14:56:29Z","abstract_excerpt":"Traditional centralized multi-agent reinforcement learning (MARL) algorithms are sometimes unpractical in complicated applications, due to non-interactivity between agents, curse of dimensionality and computation complexity. Hence, several decentralized MARL algorithms are motivated. However, existing decentralized methods only handle the fully cooperative setting where massive information needs to be transmitted in training. The block coordinate gradient descent scheme they used for successive independent actor and critic steps can simplify the calculation, but it causes serious bias. In this"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.11145","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-17T14:56:29Z","cross_cats_sorted":["cs.AI","cs.MA","stat.ML"],"title_canon_sha256":"8e8a4423b595ba3908350d2a818963d923a3271995206d2f93ca963aa0194618","abstract_canon_sha256":"68cad2d1e32278dc9c1be14aa3445f1f5853e1bf56888f16878321277629659f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:28:29.406285Z","signature_b64":"Y6ocnqsxvwSNy4tNFBI6LNYkRghsclIPE3lz0wxsb2Dy/PQfT9L4ZoUpo48047W3EobJU3mY1CnpFx54U9zcDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e045054cd4d857d06bb5016d998a9d20a785eaf66f75f1436b3fab150b69edf","last_reissued_at":"2026-07-05T06:28:29.405815Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:28:29.405815Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"F2A2: Flexible Fully-decentralized Approximate Actor-critic for Cooperative Multi-agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Bo Jin, Hongyuan Zha, Junchi Yan, Wenhao Li, Xiangfeng Wang","submitted_at":"2020-04-17T14:56:29Z","abstract_excerpt":"Traditional centralized multi-agent reinforcement learning (MARL) algorithms are sometimes unpractical in complicated applications, due to non-interactivity between agents, curse of dimensionality and computation complexity. Hence, several decentralized MARL algorithms are motivated. However, existing decentralized methods only handle the fully cooperative setting where massive information needs to be transmitted in training. The block coordinate gradient descent scheme they used for successive independent actor and critic steps can simplify the calculation, but it causes serious bias. In this"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.11145","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.11145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.11145","created_at":"2026-07-05T06:28:29.405870+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.11145v2","created_at":"2026-07-05T06:28:29.405870+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.11145","created_at":"2026-07-05T06:28:29.405870+00:00"},{"alias_kind":"pith_short_12","alias_value":"DYCFAVGNJWCX","created_at":"2026-07-05T06:28:29.405870+00:00"},{"alias_kind":"pith_short_16","alias_value":"DYCFAVGNJWCX2BV3","created_at":"2026-07-05T06:28:29.405870+00:00"},{"alias_kind":"pith_short_8","alias_value":"DYCFAVGN","created_at":"2026-07-05T06:28:29.405870+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I","json":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I.json","graph_json":"https://pith.science/api/pith-number/DYCFAVGNJWCX2BV3KALNTGFJ2I/graph.json","events_json":"https://pith.science/api/pith-number/DYCFAVGNJWCX2BV3KALNTGFJ2I/events.json","paper":"https://pith.science/paper/DYCFAVGN"},"agent_actions":{"view_html":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I","download_json":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I.json","view_paper":"https://pith.science/paper/DYCFAVGN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.11145&json=true","fetch_graph":"https://pith.science/api/pith-number/DYCFAVGNJWCX2BV3KALNTGFJ2I/graph.json","fetch_events":"https://pith.science/api/pith-number/DYCFAVGNJWCX2BV3KALNTGFJ2I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I/action/storage_attestation","attest_author":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I/action/author_attestation","sign_citation":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I/action/citation_signature","submit_replication":"https://pith.science/pith/DYCFAVGNJWCX2BV3KALNTGFJ2I/action/replication_record"}},"created_at":"2026-07-05T06:28:29.405870+00:00","updated_at":"2026-07-05T06:28:29.405870+00:00"}