{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:UPLGQ63LKNAM45GKLCYTTDBPF7","short_pith_number":"pith:UPLGQ63L","schema_version":"1.0","canonical_sha256":"a3d6687b6b5340ce74ca58b1398c2f2fdb848e75b0cd3d5f30afb205b7e5d615","source":{"kind":"arxiv","id":"1912.02288","version":2},"attestation_state":"computed","paper":{"title":"Simplified Action Decoder for Deep Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hengyuan Hu, Jakob N Foerster","submitted_at":"2019-12-04T22:34:54Z","abstract_excerpt":"In recent years we have seen fast progress on a number of benchmark problems in AI, with modern methods achieving near or super human performance in Go, Poker and Dota. One common aspect of all of these challenges is that they are by design adversarial or, technically speaking, zero-sum. In contrast to these settings, success in the real world commonly requires humans to collaborate and communicate with others, in settings that are, at least partially, cooperative. In the last year, the card game Hanabi has been established as a new benchmark environment for AI to fill this gap. In particular,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.02288","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2019-12-04T22:34:54Z","cross_cats_sorted":[],"title_canon_sha256":"b9082c76a67b3997d8fd6292758f36bf2b89360b6ac5c1b800cb11a4de3e3772","abstract_canon_sha256":"4c01eb44fbee81ffdf60500c5e4983176459cc4aa77428dfb93721890ea8c202"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:39:40.525722Z","signature_b64":"RjVTGJWb49FJ5UXrr21TqJGGPqNaDcfhcOsC9cZOVuX6+EczERySXTyZEYZRYWb0vQLwZSAthg+IbSsSMTfNAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a3d6687b6b5340ce74ca58b1398c2f2fdb848e75b0cd3d5f30afb205b7e5d615","last_reissued_at":"2026-07-05T02:39:40.525194Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:39:40.525194Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Simplified Action Decoder for Deep Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hengyuan Hu, Jakob N Foerster","submitted_at":"2019-12-04T22:34:54Z","abstract_excerpt":"In recent years we have seen fast progress on a number of benchmark problems in AI, with modern methods achieving near or super human performance in Go, Poker and Dota. One common aspect of all of these challenges is that they are by design adversarial or, technically speaking, zero-sum. In contrast to these settings, success in the real world commonly requires humans to collaborate and communicate with others, in settings that are, at least partially, cooperative. In the last year, the card game Hanabi has been established as a new benchmark environment for AI to fill this gap. In particular,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.02288","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.02288/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.02288","created_at":"2026-07-05T02:39:40.525257+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.02288v2","created_at":"2026-07-05T02:39:40.525257+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.02288","created_at":"2026-07-05T02:39:40.525257+00:00"},{"alias_kind":"pith_short_12","alias_value":"UPLGQ63LKNAM","created_at":"2026-07-05T02:39:40.525257+00:00"},{"alias_kind":"pith_short_16","alias_value":"UPLGQ63LKNAM45GK","created_at":"2026-07-05T02:39:40.525257+00:00"},{"alias_kind":"pith_short_8","alias_value":"UPLGQ63L","created_at":"2026-07-05T02:39:40.525257+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00160","citing_title":"Approximations and Learning for Decentralized Stochastic Control and Near Optimal Finite Window Policies","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7","json":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7.json","graph_json":"https://pith.science/api/pith-number/UPLGQ63LKNAM45GKLCYTTDBPF7/graph.json","events_json":"https://pith.science/api/pith-number/UPLGQ63LKNAM45GKLCYTTDBPF7/events.json","paper":"https://pith.science/paper/UPLGQ63L"},"agent_actions":{"view_html":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7","download_json":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7.json","view_paper":"https://pith.science/paper/UPLGQ63L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.02288&json=true","fetch_graph":"https://pith.science/api/pith-number/UPLGQ63LKNAM45GKLCYTTDBPF7/graph.json","fetch_events":"https://pith.science/api/pith-number/UPLGQ63LKNAM45GKLCYTTDBPF7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7/action/storage_attestation","attest_author":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7/action/author_attestation","sign_citation":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7/action/citation_signature","submit_replication":"https://pith.science/pith/UPLGQ63LKNAM45GKLCYTTDBPF7/action/replication_record"}},"created_at":"2026-07-05T02:39:40.525257+00:00","updated_at":"2026-07-05T02:39:40.525257+00:00"}