{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WXFRDHNO5PSRJ7MKIUQMQ7KUZB","short_pith_number":"pith:WXFRDHNO","schema_version":"1.0","canonical_sha256":"b5cb119daeebe514fd8a4520c87d54c87f3af5e25dc819fed46192794b89de3c","source":{"kind":"arxiv","id":"2311.06210","version":1},"attestation_state":"computed","paper":{"title":"Optimal Cooperative Multiplayer Learning Bandits with Noisy Rewards and No Communication","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"William Chang, Yuanhao Lu","submitted_at":"2023-11-10T17:55:44Z","abstract_excerpt":"We consider a cooperative multiplayer bandit learning problem where the players are only allowed to agree on a strategy beforehand, but cannot communicate during the learning process. In this problem, each player simultaneously selects an action. Based on the actions selected by all players, the team of players receives a reward. The actions of all the players are commonly observed. However, each player receives a noisy version of the reward which cannot be shared with other players. Since players receive potentially different rewards, there is an asymmetry in the information used to select th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.06210","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2023-11-10T17:55:44Z","cross_cats_sorted":["cs.MA","stat.ML"],"title_canon_sha256":"ad7ee91e4c5c3c79729e2253e2cc7e517258fc562d9335c99835c17859455ff1","abstract_canon_sha256":"e89c42d0ee9761ec6eb206e4f9e8d8de4376122d87e52d794f2dce245e23aac8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:29.791578Z","signature_b64":"fEd9/7Un4bRB4hUS/7bi+AUtP9VGG05ezrU0H6syVyUgDImtDKpS0q6srLxEwS7Ng3rYPv0CCTdPu2wPD89jCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b5cb119daeebe514fd8a4520c87d54c87f3af5e25dc819fed46192794b89de3c","last_reissued_at":"2026-07-05T07:11:29.791168Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:29.791168Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimal Cooperative Multiplayer Learning Bandits with Noisy Rewards and No Communication","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"William Chang, Yuanhao Lu","submitted_at":"2023-11-10T17:55:44Z","abstract_excerpt":"We consider a cooperative multiplayer bandit learning problem where the players are only allowed to agree on a strategy beforehand, but cannot communicate during the learning process. In this problem, each player simultaneously selects an action. Based on the actions selected by all players, the team of players receives a reward. The actions of all the players are commonly observed. However, each player receives a noisy version of the reward which cannot be shared with other players. Since players receive potentially different rewards, there is an asymmetry in the information used to select th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.06210","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.06210/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.06210","created_at":"2026-07-05T07:11:29.791217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.06210v1","created_at":"2026-07-05T07:11:29.791217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.06210","created_at":"2026-07-05T07:11:29.791217+00:00"},{"alias_kind":"pith_short_12","alias_value":"WXFRDHNO5PSR","created_at":"2026-07-05T07:11:29.791217+00:00"},{"alias_kind":"pith_short_16","alias_value":"WXFRDHNO5PSRJ7MK","created_at":"2026-07-05T07:11:29.791217+00:00"},{"alias_kind":"pith_short_8","alias_value":"WXFRDHNO","created_at":"2026-07-05T07:11:29.791217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.10529","citing_title":"Robust Multi-Agent Bandits with Heavy-Tailed Rewards and Information Asymmetry","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB","json":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB.json","graph_json":"https://pith.science/api/pith-number/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/graph.json","events_json":"https://pith.science/api/pith-number/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/events.json","paper":"https://pith.science/paper/WXFRDHNO"},"agent_actions":{"view_html":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB","download_json":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB.json","view_paper":"https://pith.science/paper/WXFRDHNO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.06210&json=true","fetch_graph":"https://pith.science/api/pith-number/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/graph.json","fetch_events":"https://pith.science/api/pith-number/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/action/storage_attestation","attest_author":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/action/author_attestation","sign_citation":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/action/citation_signature","submit_replication":"https://pith.science/pith/WXFRDHNO5PSRJ7MKIUQMQ7KUZB/action/replication_record"}},"created_at":"2026-07-05T07:11:29.791217+00:00","updated_at":"2026-07-05T07:11:29.791217+00:00"}