{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:VKXWDGEISIRTB7LPMBBIPQYMQT","short_pith_number":"pith:VKXWDGEI","schema_version":"1.0","canonical_sha256":"aaaf619888922330fd6f604287c30c84d3ba402e7c6f9832b46437ac6181a9af","source":{"kind":"arxiv","id":"2112.11701","version":3},"attestation_state":"computed","paper":{"title":"Maximum Entropy Population-Based Training for Zero-Shot Human-AI Coordination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hu Haifeng, Jinming Song, Rui Zhao, Yang Gao, Yang Wei, Yi Wu, Yufeng Yuan, Zhongqian Sun","submitted_at":"2021-12-22T07:19:36Z","abstract_excerpt":"We study the problem of training a Reinforcement Learning (RL) agent that is collaborative with humans without using any human data. Although such agents can be obtained through self-play training, they can suffer significantly from distributional shift when paired with unencountered partners, such as humans. To mitigate this distributional shift, we propose Maximum Entropy Population-based training (MEP). In MEP, agents in the population are trained with our derived Population Entropy bonus to promote both pairwise diversity between agents and individual diversity of agents themselves, and a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.11701","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-12-22T07:19:36Z","cross_cats_sorted":[],"title_canon_sha256":"8b11643d1e95168a9b6c4e3d7bde331def34425d0ad61f322edcb39e701cf81d","abstract_canon_sha256":"d6a405d9e15a998a8442f85701deb4e6a74ad63b489c96e8f4aeb820cfb5ca79"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:34:49.189045Z","signature_b64":"09VeD6JHmcjlbnQlszTuC5UDZiCFZm29WjFR5+Vl4xgav3dRCMBJZ1Jqhzd3JrgijAKpw6hXpFd32N1SMFmGBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aaaf619888922330fd6f604287c30c84d3ba402e7c6f9832b46437ac6181a9af","last_reissued_at":"2026-07-05T04:34:49.188543Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:34:49.188543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Maximum Entropy Population-Based Training for Zero-Shot Human-AI Coordination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hu Haifeng, Jinming Song, Rui Zhao, Yang Gao, Yang Wei, Yi Wu, Yufeng Yuan, Zhongqian Sun","submitted_at":"2021-12-22T07:19:36Z","abstract_excerpt":"We study the problem of training a Reinforcement Learning (RL) agent that is collaborative with humans without using any human data. Although such agents can be obtained through self-play training, they can suffer significantly from distributional shift when paired with unencountered partners, such as humans. To mitigate this distributional shift, we propose Maximum Entropy Population-based training (MEP). In MEP, agents in the population are trained with our derived Population Entropy bonus to promote both pairwise diversity between agents and individual diversity of agents themselves, and a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.11701","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.11701/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.11701","created_at":"2026-07-05T04:34:49.188607+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.11701v3","created_at":"2026-07-05T04:34:49.188607+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.11701","created_at":"2026-07-05T04:34:49.188607+00:00"},{"alias_kind":"pith_short_12","alias_value":"VKXWDGEISIRT","created_at":"2026-07-05T04:34:49.188607+00:00"},{"alias_kind":"pith_short_16","alias_value":"VKXWDGEISIRTB7LP","created_at":"2026-07-05T04:34:49.188607+00:00"},{"alias_kind":"pith_short_8","alias_value":"VKXWDGEI","created_at":"2026-07-05T04:34:49.188607+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.01336","citing_title":"Enhancing Diversity in Parallel Agents: A Maximum State Entropy Exploration Story","ref_index":69,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT","json":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT.json","graph_json":"https://pith.science/api/pith-number/VKXWDGEISIRTB7LPMBBIPQYMQT/graph.json","events_json":"https://pith.science/api/pith-number/VKXWDGEISIRTB7LPMBBIPQYMQT/events.json","paper":"https://pith.science/paper/VKXWDGEI"},"agent_actions":{"view_html":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT","download_json":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT.json","view_paper":"https://pith.science/paper/VKXWDGEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.11701&json=true","fetch_graph":"https://pith.science/api/pith-number/VKXWDGEISIRTB7LPMBBIPQYMQT/graph.json","fetch_events":"https://pith.science/api/pith-number/VKXWDGEISIRTB7LPMBBIPQYMQT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT/action/storage_attestation","attest_author":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT/action/author_attestation","sign_citation":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT/action/citation_signature","submit_replication":"https://pith.science/pith/VKXWDGEISIRTB7LPMBBIPQYMQT/action/replication_record"}},"created_at":"2026-07-05T04:34:49.188607+00:00","updated_at":"2026-07-05T04:34:49.188607+00:00"}