{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:JZ5ONMED3FUYNEYKV4TB32BNJE","short_pith_number":"pith:JZ5ONMED","schema_version":"1.0","canonical_sha256":"4e7ae6b083d96986930aaf261de82d4922527c0d382b305a806a6fbb575b0844","source":{"kind":"arxiv","id":"2108.01832","version":2},"attestation_state":"computed","paper":{"title":"Offline Decentralized Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Jiechuan Jiang, Zongqing Lu","submitted_at":"2021-08-04T03:53:33Z","abstract_excerpt":"In many real-world multi-agent cooperative tasks, due to high cost and risk, agents cannot continuously interact with the environment and collect experiences during learning, but have to learn from offline datasets. However, the transition dynamics in the dataset of each agent can be much different from the ones induced by the learned policies of other agents in execution, creating large errors in value estimates. Consequently, agents learn uncoordinated low-performing policies. In this paper, we propose a framework for offline decentralized multi-agent reinforcement learning, which exploits v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.01832","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-08-04T03:53:33Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"2f352491787acff2d7cdc954325aac707e44bb71712613659ca330bcb4ebe015","abstract_canon_sha256":"c942a37202f47e4d4d2f08931be63767981d1ef766c24fa770e51f5ea64ff7f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:42.212779Z","signature_b64":"b93tKcxH7y64jTI/mxARwxuaPVcckYEORFHc2m8dsv4yNtqQu4VKwRe5FjTOX8DvuC5u6wPhVt/MBminnt9yDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e7ae6b083d96986930aaf261de82d4922527c0d382b305a806a6fbb575b0844","last_reissued_at":"2026-07-05T06:35:42.212256Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:42.212256Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Decentralized Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Jiechuan Jiang, Zongqing Lu","submitted_at":"2021-08-04T03:53:33Z","abstract_excerpt":"In many real-world multi-agent cooperative tasks, due to high cost and risk, agents cannot continuously interact with the environment and collect experiences during learning, but have to learn from offline datasets. However, the transition dynamics in the dataset of each agent can be much different from the ones induced by the learned policies of other agents in execution, creating large errors in value estimates. Consequently, agents learn uncoordinated low-performing policies. In this paper, we propose a framework for offline decentralized multi-agent reinforcement learning, which exploits v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.01832","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.01832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.01832","created_at":"2026-07-05T06:35:42.212323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.01832v2","created_at":"2026-07-05T06:35:42.212323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.01832","created_at":"2026-07-05T06:35:42.212323+00:00"},{"alias_kind":"pith_short_12","alias_value":"JZ5ONMED3FUY","created_at":"2026-07-05T06:35:42.212323+00:00"},{"alias_kind":"pith_short_16","alias_value":"JZ5ONMED3FUYNEYK","created_at":"2026-07-05T06:35:42.212323+00:00"},{"alias_kind":"pith_short_8","alias_value":"JZ5ONMED","created_at":"2026-07-05T06:35:42.212323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02496","citing_title":"Controllable Sim Agents with Behavior Latents","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23308","citing_title":"CODA: Coordination via On-Policy Diffusion for Multi-Agent Offline Reinforcement Learning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE","json":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE.json","graph_json":"https://pith.science/api/pith-number/JZ5ONMED3FUYNEYKV4TB32BNJE/graph.json","events_json":"https://pith.science/api/pith-number/JZ5ONMED3FUYNEYKV4TB32BNJE/events.json","paper":"https://pith.science/paper/JZ5ONMED"},"agent_actions":{"view_html":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE","download_json":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE.json","view_paper":"https://pith.science/paper/JZ5ONMED","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.01832&json=true","fetch_graph":"https://pith.science/api/pith-number/JZ5ONMED3FUYNEYKV4TB32BNJE/graph.json","fetch_events":"https://pith.science/api/pith-number/JZ5ONMED3FUYNEYKV4TB32BNJE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE/action/storage_attestation","attest_author":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE/action/author_attestation","sign_citation":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE/action/citation_signature","submit_replication":"https://pith.science/pith/JZ5ONMED3FUYNEYKV4TB32BNJE/action/replication_record"}},"created_at":"2026-07-05T06:35:42.212323+00:00","updated_at":"2026-07-05T06:35:42.212323+00:00"}