{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XPOR4RDWQC7UVAP64N4MWDPU5Z","short_pith_number":"pith:XPOR4RDW","schema_version":"1.0","canonical_sha256":"bbdd1e447680bf4a81fee378cb0df4ee7f5d3a9be9ddb7aa7988e7161c577cc8","source":{"kind":"arxiv","id":"2111.03431","version":1},"attestation_state":"computed","paper":{"title":"Learning to Cooperate with Unseen Agent via Meta-Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Nat Dilokthanakul, Poramate Manoonpong, Rujikorn Charakorn","submitted_at":"2021-11-05T12:01:28Z","abstract_excerpt":"Ad hoc teamwork problem describes situations where an agent has to cooperate with previously unseen agents to achieve a common goal. For an agent to be successful in these scenarios, it has to have a suitable cooperative skill. One could implement cooperative skills into an agent by using domain knowledge to design the agent's behavior. However, in complex domains, domain knowledge might not be available. Therefore, it is worthwhile to explore how to directly learn cooperative skills from data. In this work, we apply meta-reinforcement learning (meta-RL) formulation in the context of the ad ho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.03431","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-11-05T12:01:28Z","cross_cats_sorted":["cs.LG","cs.MA"],"title_canon_sha256":"9a32b090bf67e1390b355d44d01d5c4c8e6dd1e1dfba6cbf657254b50e7e1b59","abstract_canon_sha256":"9ecfc844e1caa5c4f5dfcee559ee4a301496f4977141862490cdcf3c887b5cc4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:08:25.501916Z","signature_b64":"lxkGRBmSeQ/Jq0NctrWJpaHj6Fc8OWTahICIFkNzuJAZTPqvTTa6/NS6sIsy7Y2ofohySlhQ1QxRp47IS05dBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bbdd1e447680bf4a81fee378cb0df4ee7f5d3a9be9ddb7aa7988e7161c577cc8","last_reissued_at":"2026-07-05T05:08:25.501436Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:08:25.501436Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Cooperate with Unseen Agent via Meta-Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA"],"primary_cat":"cs.AI","authors_text":"Nat Dilokthanakul, Poramate Manoonpong, Rujikorn Charakorn","submitted_at":"2021-11-05T12:01:28Z","abstract_excerpt":"Ad hoc teamwork problem describes situations where an agent has to cooperate with previously unseen agents to achieve a common goal. For an agent to be successful in these scenarios, it has to have a suitable cooperative skill. One could implement cooperative skills into an agent by using domain knowledge to design the agent's behavior. However, in complex domains, domain knowledge might not be available. Therefore, it is worthwhile to explore how to directly learn cooperative skills from data. In this work, we apply meta-reinforcement learning (meta-RL) formulation in the context of the ad ho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.03431","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.03431/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.03431","created_at":"2026-07-05T05:08:25.501506+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.03431v1","created_at":"2026-07-05T05:08:25.501506+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.03431","created_at":"2026-07-05T05:08:25.501506+00:00"},{"alias_kind":"pith_short_12","alias_value":"XPOR4RDWQC7U","created_at":"2026-07-05T05:08:25.501506+00:00"},{"alias_kind":"pith_short_16","alias_value":"XPOR4RDWQC7UVAP6","created_at":"2026-07-05T05:08:25.501506+00:00"},{"alias_kind":"pith_short_8","alias_value":"XPOR4RDW","created_at":"2026-07-05T05:08:25.501506+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.07978","citing_title":"A Survey of In-Context Reinforcement Learning","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z","json":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z.json","graph_json":"https://pith.science/api/pith-number/XPOR4RDWQC7UVAP64N4MWDPU5Z/graph.json","events_json":"https://pith.science/api/pith-number/XPOR4RDWQC7UVAP64N4MWDPU5Z/events.json","paper":"https://pith.science/paper/XPOR4RDW"},"agent_actions":{"view_html":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z","download_json":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z.json","view_paper":"https://pith.science/paper/XPOR4RDW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.03431&json=true","fetch_graph":"https://pith.science/api/pith-number/XPOR4RDWQC7UVAP64N4MWDPU5Z/graph.json","fetch_events":"https://pith.science/api/pith-number/XPOR4RDWQC7UVAP64N4MWDPU5Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z/action/storage_attestation","attest_author":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z/action/author_attestation","sign_citation":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z/action/citation_signature","submit_replication":"https://pith.science/pith/XPOR4RDWQC7UVAP64N4MWDPU5Z/action/replication_record"}},"created_at":"2026-07-05T05:08:25.501506+00:00","updated_at":"2026-07-05T05:08:25.501506+00:00"}