{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:2U4K4GDQ3Q6TL7CUWYJGRGVDRN","short_pith_number":"pith:2U4K4GDQ","canonical_record":{"source":{"id":"2302.01605","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-02-03T09:06:42Z","cross_cats_sorted":[],"title_canon_sha256":"2a1cd9eea1ea3127f14e6feb1e0aa8e4ca730f8d726a31d4c1b518585b0fa548","abstract_canon_sha256":"c8eca29e271ee84332cd74c17ffe90acd9c5f4974d3a187c047fa447be905026"},"schema_version":"1.0"},"canonical_sha256":"d538ae1870dc3d35fc54b612689aa38b71a70fb6c97dd2d627667948f30098f3","source":{"kind":"arxiv","id":"2302.01605","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.01605","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"arxiv_version","alias_value":"2302.01605v1","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.01605","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"pith_short_12","alias_value":"2U4K4GDQ3Q6T","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"pith_short_16","alias_value":"2U4K4GDQ3Q6TL7CU","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"pith_short_8","alias_value":"2U4K4GDQ","created_at":"2026-07-05T05:38:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:2U4K4GDQ3Q6TL7CUWYJGRGVDRN","target":"record","payload":{"canonical_record":{"source":{"id":"2302.01605","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-02-03T09:06:42Z","cross_cats_sorted":[],"title_canon_sha256":"2a1cd9eea1ea3127f14e6feb1e0aa8e4ca730f8d726a31d4c1b518585b0fa548","abstract_canon_sha256":"c8eca29e271ee84332cd74c17ffe90acd9c5f4974d3a187c047fa447be905026"},"schema_version":"1.0"},"canonical_sha256":"d538ae1870dc3d35fc54b612689aa38b71a70fb6c97dd2d627667948f30098f3","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:38:34.335875Z","signature_b64":"vlIGKzQb/AeMvxWbDAlDqXgvDHfknh2wcz+CmdtmN4D7mGWiOypiCAt3oL/dCrQ9bwaUZqJn87jwowvVOCHtDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d538ae1870dc3d35fc54b612689aa38b71a70fb6c97dd2d627667948f30098f3","last_reissued_at":"2026-07-05T05:38:34.335415Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:38:34.335415Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2302.01605","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:38:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JXdoUIZyuUxBsEsD0F8IfAowY+O/UX83/KR6HpMznGaL74r1SF5Qcuv2wOmKLPL18FP03SmcOyQFOJ62UcbFBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T15:50:04.143818Z"},"content_sha256":"f046a61670bd151e0a364658d9037b37ccf3c6df4397a9de7bc1936c89e45cb6","schema_version":"1.0","event_id":"sha256:f046a61670bd151e0a364658d9037b37ccf3c6df4397a9de7bc1936c89e45cb6"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:2U4K4GDQ3Q6TL7CUWYJGRGVDRN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Learning Zero-Shot Cooperation with Humans, Assuming Humans Are Biased","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Botian Xu, Chao Yu, Hao Tang, Jiaqi Yang, Jiaxuan Gao, Weilin Liu, Yi Wu, Yu Wang","submitted_at":"2023-02-03T09:06:42Z","abstract_excerpt":"There is a recent trend of applying multi-agent reinforcement learning (MARL) to train an agent that can cooperate with humans in a zero-shot fashion without using any human data. The typical workflow is to first repeatedly run self-play (SP) to build a policy pool and then train the final adaptive policy against this pool. A crucial limitation of this framework is that every policy in the pool is optimized w.r.t. the environment reward function, which implicitly assumes that the testing partners of the adaptive policy will be precisely optimizing the same reward function as well. However, hum"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.01605","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.01605/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:38:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9oXbxVHtW2CPMd/oVqCFhkiIsMb48Ru1JG6OkAoqymjcEX/I6o7XMoAjgeGCCcqdtfL37waZeDfT4yhhmfU1CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T15:50:04.144192Z"},"content_sha256":"6c3a504aa31d8aeafd1cd5f0b1d0cb47a04b398ef273f04322c38fc3ad574b45","schema_version":"1.0","event_id":"sha256:6c3a504aa31d8aeafd1cd5f0b1d0cb47a04b398ef273f04322c38fc3ad574b45"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2U4K4GDQ3Q6TL7CUWYJGRGVDRN/bundle.json","state_url":"https://pith.science/pith/2U4K4GDQ3Q6TL7CUWYJGRGVDRN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2U4K4GDQ3Q6TL7CUWYJGRGVDRN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-22T15:50:04Z","links":{"resolver":"https://pith.science/pith/2U4K4GDQ3Q6TL7CUWYJGRGVDRN","bundle":"https://pith.science/pith/2U4K4GDQ3Q6TL7CUWYJGRGVDRN/bundle.json","state":"https://pith.science/pith/2U4K4GDQ3Q6TL7CUWYJGRGVDRN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2U4K4GDQ3Q6TL7CUWYJGRGVDRN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:2U4K4GDQ3Q6TL7CUWYJGRGVDRN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c8eca29e271ee84332cd74c17ffe90acd9c5f4974d3a187c047fa447be905026","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-02-03T09:06:42Z","title_canon_sha256":"2a1cd9eea1ea3127f14e6feb1e0aa8e4ca730f8d726a31d4c1b518585b0fa548"},"schema_version":"1.0","source":{"id":"2302.01605","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2302.01605","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"arxiv_version","alias_value":"2302.01605v1","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.01605","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"pith_short_12","alias_value":"2U4K4GDQ3Q6T","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"pith_short_16","alias_value":"2U4K4GDQ3Q6TL7CU","created_at":"2026-07-05T05:38:34Z"},{"alias_kind":"pith_short_8","alias_value":"2U4K4GDQ","created_at":"2026-07-05T05:38:34Z"}],"graph_snapshots":[{"event_id":"sha256:6c3a504aa31d8aeafd1cd5f0b1d0cb47a04b398ef273f04322c38fc3ad574b45","target":"graph","created_at":"2026-07-05T05:38:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2302.01605/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"There is a recent trend of applying multi-agent reinforcement learning (MARL) to train an agent that can cooperate with humans in a zero-shot fashion without using any human data. The typical workflow is to first repeatedly run self-play (SP) to build a policy pool and then train the final adaptive policy against this pool. A crucial limitation of this framework is that every policy in the pool is optimized w.r.t. the environment reward function, which implicitly assumes that the testing partners of the adaptive policy will be precisely optimizing the same reward function as well. However, hum","authors_text":"Botian Xu, Chao Yu, Hao Tang, Jiaqi Yang, Jiaxuan Gao, Weilin Liu, Yi Wu, Yu Wang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-02-03T09:06:42Z","title":"Learning Zero-Shot Cooperation with Humans, Assuming Humans Are Biased"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.01605","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f046a61670bd151e0a364658d9037b37ccf3c6df4397a9de7bc1936c89e45cb6","target":"record","created_at":"2026-07-05T05:38:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c8eca29e271ee84332cd74c17ffe90acd9c5f4974d3a187c047fa447be905026","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-02-03T09:06:42Z","title_canon_sha256":"2a1cd9eea1ea3127f14e6feb1e0aa8e4ca730f8d726a31d4c1b518585b0fa548"},"schema_version":"1.0","source":{"id":"2302.01605","kind":"arxiv","version":1}},"canonical_sha256":"d538ae1870dc3d35fc54b612689aa38b71a70fb6c97dd2d627667948f30098f3","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d538ae1870dc3d35fc54b612689aa38b71a70fb6c97dd2d627667948f30098f3","first_computed_at":"2026-07-05T05:38:34.335415Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:38:34.335415Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"vlIGKzQb/AeMvxWbDAlDqXgvDHfknh2wcz+CmdtmN4D7mGWiOypiCAt3oL/dCrQ9bwaUZqJn87jwowvVOCHtDg==","signature_status":"signed_v1","signed_at":"2026-07-05T05:38:34.335875Z","signed_message":"canonical_sha256_bytes"},"source_id":"2302.01605","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f046a61670bd151e0a364658d9037b37ccf3c6df4397a9de7bc1936c89e45cb6","sha256:6c3a504aa31d8aeafd1cd5f0b1d0cb47a04b398ef273f04322c38fc3ad574b45"],"state_sha256":"d73e08d1b40f4c8d7963cecd04c248f4266176ab39277de3bcb5dcf3401a33ea"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SuuTXO5Npe0BhHQLUd716EGOQCaFkgdzDP2Jozd17CCvLQgvkx4lSNmrdVqzde6F0q1SSSaIP8vdjJVYSSuODg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-22T15:50:04.146390Z","bundle_sha256":"262d1dfae36ef9dd4df83b05f7d055d2937acfc1f68865aed3871cc4675d3f34"}}