{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:WEZNAEZKNBYFJ7YSL7XC2TIAF7","short_pith_number":"pith:WEZNAEZK","schema_version":"1.0","canonical_sha256":"b132d0132a687054ff125fee2d4d002fed21bd5884503d5bf63418f9a50dfa4d","source":{"kind":"arxiv","id":"2102.02274","version":1},"attestation_state":"computed","paper":{"title":"Neural Recursive Belief States in Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Bernardo Avila Pires, Edward Hughes, Kevin R. McKee, Pol Moreno, Th\\'eophane Weber","submitted_at":"2021-02-03T20:10:23Z","abstract_excerpt":"In multi-agent reinforcement learning, the problem of learning to act is particularly difficult because the policies of co-players may be heavily conditioned on information only observed by them. On the other hand, humans readily form beliefs about the knowledge possessed by their peers and leverage beliefs to inform decision-making. Such abilities underlie individual success in a wide range of Markov games, from bluffing in Poker to conditional cooperation in the Prisoner's Dilemma, to convention-building in Bridge. Classical methods are usually not applicable to complex domains due to the in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.02274","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-03T20:10:23Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"01d95ede9c7681b759eea7be09109280f536423acd1f436405804d148be922c6","abstract_canon_sha256":"597d2d14e52c6e72ebb0e36dd8206cb7bb03f77c6c770f5d8bc2fa88ec4e3651"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:12:46.616964Z","signature_b64":"pQgtd5Kqx8EV9j7CQB97GuCVeB/lyBiNaLTVZms4j5WKCEQuZMs8JbTzQQx+PBCbNwJSy8z8GsaqKtm8F7zLCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b132d0132a687054ff125fee2d4d002fed21bd5884503d5bf63418f9a50dfa4d","last_reissued_at":"2026-07-05T02:12:46.616539Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:12:46.616539Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neural Recursive Belief States in Multi-Agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.LG","authors_text":"Bernardo Avila Pires, Edward Hughes, Kevin R. McKee, Pol Moreno, Th\\'eophane Weber","submitted_at":"2021-02-03T20:10:23Z","abstract_excerpt":"In multi-agent reinforcement learning, the problem of learning to act is particularly difficult because the policies of co-players may be heavily conditioned on information only observed by them. On the other hand, humans readily form beliefs about the knowledge possessed by their peers and leverage beliefs to inform decision-making. Such abilities underlie individual success in a wide range of Markov games, from bluffing in Poker to conditional cooperation in the Prisoner's Dilemma, to convention-building in Bridge. Classical methods are usually not applicable to complex domains due to the in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.02274","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.02274/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.02274","created_at":"2026-07-05T02:12:46.616600+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.02274v1","created_at":"2026-07-05T02:12:46.616600+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.02274","created_at":"2026-07-05T02:12:46.616600+00:00"},{"alias_kind":"pith_short_12","alias_value":"WEZNAEZKNBYF","created_at":"2026-07-05T02:12:46.616600+00:00"},{"alias_kind":"pith_short_16","alias_value":"WEZNAEZKNBYFJ7YS","created_at":"2026-07-05T02:12:46.616600+00:00"},{"alias_kind":"pith_short_8","alias_value":"WEZNAEZK","created_at":"2026-07-05T02:12:46.616600+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14121","citing_title":"An Encoded Corrective Double Deep Q-Networks for Multi-Agent Control Systems","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7","json":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7.json","graph_json":"https://pith.science/api/pith-number/WEZNAEZKNBYFJ7YSL7XC2TIAF7/graph.json","events_json":"https://pith.science/api/pith-number/WEZNAEZKNBYFJ7YSL7XC2TIAF7/events.json","paper":"https://pith.science/paper/WEZNAEZK"},"agent_actions":{"view_html":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7","download_json":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7.json","view_paper":"https://pith.science/paper/WEZNAEZK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.02274&json=true","fetch_graph":"https://pith.science/api/pith-number/WEZNAEZKNBYFJ7YSL7XC2TIAF7/graph.json","fetch_events":"https://pith.science/api/pith-number/WEZNAEZKNBYFJ7YSL7XC2TIAF7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7/action/storage_attestation","attest_author":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7/action/author_attestation","sign_citation":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7/action/citation_signature","submit_replication":"https://pith.science/pith/WEZNAEZKNBYFJ7YSL7XC2TIAF7/action/replication_record"}},"created_at":"2026-07-05T02:12:46.616600+00:00","updated_at":"2026-07-05T02:12:46.616600+00:00"}