{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:BKOQUSAC44UHJDNYIK5IE6EH5T","short_pith_number":"pith:BKOQUSAC","schema_version":"1.0","canonical_sha256":"0a9d0a4802e728748db842ba827887ecf75a55ee1a83343265bd912492a20de1","source":{"kind":"arxiv","id":"2101.03864","version":2},"attestation_state":"computed","paper":{"title":"Deep Interactive Bayesian Reinforcement Learning via Meta-Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Kamil Ciosek, Katja Hofmann, Luisa Zintgraf, Sam Devlin, Shimon Whiteson","submitted_at":"2021-01-11T13:25:13Z","abstract_excerpt":"Agents that interact with other agents often do not know a priori what the other agents' strategies are, but have to maximise their own online return while interacting with and learning about others. The optimal adaptive behaviour under uncertainty over the other agents' strategies w.r.t. some prior can in principle be computed using the Interactive Bayesian Reinforcement Learning framework. Unfortunately, doing so is intractable in most settings, and existing approximation methods are restricted to small tasks. To overcome this, we propose to meta-learn approximate belief inference and Bayes-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.03864","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-11T13:25:13Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"2ac0dc1b7a3571a7f0a220ca8a8375e03aac36a7c9ffbced35492a1cd7c9058a","abstract_canon_sha256":"e0da1b28ec38f938cc20004c80d9190ae590a8cbd1fc538ea6bc83e5974e05ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:15:07.788543Z","signature_b64":"wXiC0CgekAyK3xjbulMngnI1xFkPc5unAEg2KNTlqDpDb+okEFDaL9wvp5/qDvCF7zT5QoblzudsbgLxeQrWBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a9d0a4802e728748db842ba827887ecf75a55ee1a83343265bd912492a20de1","last_reissued_at":"2026-07-05T04:15:07.788060Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:15:07.788060Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Interactive Bayesian Reinforcement Learning via Meta-Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Kamil Ciosek, Katja Hofmann, Luisa Zintgraf, Sam Devlin, Shimon Whiteson","submitted_at":"2021-01-11T13:25:13Z","abstract_excerpt":"Agents that interact with other agents often do not know a priori what the other agents' strategies are, but have to maximise their own online return while interacting with and learning about others. The optimal adaptive behaviour under uncertainty over the other agents' strategies w.r.t. some prior can in principle be computed using the Interactive Bayesian Reinforcement Learning framework. Unfortunately, doing so is intractable in most settings, and existing approximation methods are restricted to small tasks. To overcome this, we propose to meta-learn approximate belief inference and Bayes-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.03864","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.03864/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.03864","created_at":"2026-07-05T04:15:07.788116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.03864v2","created_at":"2026-07-05T04:15:07.788116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.03864","created_at":"2026-07-05T04:15:07.788116+00:00"},{"alias_kind":"pith_short_12","alias_value":"BKOQUSAC44UH","created_at":"2026-07-05T04:15:07.788116+00:00"},{"alias_kind":"pith_short_16","alias_value":"BKOQUSAC44UHJDNY","created_at":"2026-07-05T04:15:07.788116+00:00"},{"alias_kind":"pith_short_8","alias_value":"BKOQUSAC","created_at":"2026-07-05T04:15:07.788116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07301","citing_title":"SOM: Structured Opponent Modeling for LLM-based Agents via Structural Causal Model","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T","json":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T.json","graph_json":"https://pith.science/api/pith-number/BKOQUSAC44UHJDNYIK5IE6EH5T/graph.json","events_json":"https://pith.science/api/pith-number/BKOQUSAC44UHJDNYIK5IE6EH5T/events.json","paper":"https://pith.science/paper/BKOQUSAC"},"agent_actions":{"view_html":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T","download_json":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T.json","view_paper":"https://pith.science/paper/BKOQUSAC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.03864&json=true","fetch_graph":"https://pith.science/api/pith-number/BKOQUSAC44UHJDNYIK5IE6EH5T/graph.json","fetch_events":"https://pith.science/api/pith-number/BKOQUSAC44UHJDNYIK5IE6EH5T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T/action/storage_attestation","attest_author":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T/action/author_attestation","sign_citation":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T/action/citation_signature","submit_replication":"https://pith.science/pith/BKOQUSAC44UHJDNYIK5IE6EH5T/action/replication_record"}},"created_at":"2026-07-05T04:15:07.788116+00:00","updated_at":"2026-07-05T04:15:07.788116+00:00"}