{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YXZWKUANA4D4R6DRXE4DEGWT5C","short_pith_number":"pith:YXZWKUAN","schema_version":"1.0","canonical_sha256":"c5f365500d0707c8f871b938321ad3e888e2467e5de29a8f7ac38c3ad3181a89","source":{"kind":"arxiv","id":"2502.17625","version":1},"attestation_state":"computed","paper":{"title":"Instance-Dependent Regret Bounds for Learning Two-Player Zero-Sum Games with Bandit Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.LG","authors_text":"Haipeng Luo, Shinji Ito, Taira Tsuchiya, Yue Wu","submitted_at":"2025-02-24T20:20:06Z","abstract_excerpt":"No-regret self-play learning dynamics have become one of the premier ways to solve large-scale games in practice. Accelerating their convergence via improving the regret of the players over the naive $O(\\sqrt{T})$ bound after $T$ rounds has been extensively studied in recent years, but almost all studies assume access to exact gradient feedback. We address the question of whether acceleration is possible under bandit feedback only and provide an affirmative answer for two-player zero-sum normal-form games. Specifically, we show that if both players apply the Tsallis-INF algorithm of Zimmert an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17625","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-24T20:20:06Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"920b554cb4efe45bc73c1d68e32cf8db56684112e1f1d673a1ec3d107d549b4d","abstract_canon_sha256":"d22f91ce86f55025171dc84a1b76a34bf98374c9071ddb9c0865730e798153dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:41.375504Z","signature_b64":"5cxHHeOl9qCR1+0GMoKCcH1iyA8WranexZdrwnY/N1aRFUwfSnjNezJsiz1V6GqmYJhmkkk14/SW9hG0BpwnBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5f365500d0707c8f871b938321ad3e888e2467e5de29a8f7ac38c3ad3181a89","last_reissued_at":"2026-07-05T10:19:41.375082Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:41.375082Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Instance-Dependent Regret Bounds for Learning Two-Player Zero-Sum Games with Bandit Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.LG","authors_text":"Haipeng Luo, Shinji Ito, Taira Tsuchiya, Yue Wu","submitted_at":"2025-02-24T20:20:06Z","abstract_excerpt":"No-regret self-play learning dynamics have become one of the premier ways to solve large-scale games in practice. Accelerating their convergence via improving the regret of the players over the naive $O(\\sqrt{T})$ bound after $T$ rounds has been extensively studied in recent years, but almost all studies assume access to exact gradient feedback. We address the question of whether acceleration is possible under bandit feedback only and provide an affirmative answer for two-player zero-sum normal-form games. Specifically, we show that if both players apply the Tsallis-INF algorithm of Zimmert an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17625","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17625/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17625","created_at":"2026-07-05T10:19:41.375140+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17625v1","created_at":"2026-07-05T10:19:41.375140+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17625","created_at":"2026-07-05T10:19:41.375140+00:00"},{"alias_kind":"pith_short_12","alias_value":"YXZWKUANA4D4","created_at":"2026-07-05T10:19:41.375140+00:00"},{"alias_kind":"pith_short_16","alias_value":"YXZWKUANA4D4R6DR","created_at":"2026-07-05T10:19:41.375140+00:00"},{"alias_kind":"pith_short_8","alias_value":"YXZWKUAN","created_at":"2026-07-05T10:19:41.375140+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.03802","citing_title":"Learning in Matching Games with Bandit Feedback","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C","json":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C.json","graph_json":"https://pith.science/api/pith-number/YXZWKUANA4D4R6DRXE4DEGWT5C/graph.json","events_json":"https://pith.science/api/pith-number/YXZWKUANA4D4R6DRXE4DEGWT5C/events.json","paper":"https://pith.science/paper/YXZWKUAN"},"agent_actions":{"view_html":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C","download_json":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C.json","view_paper":"https://pith.science/paper/YXZWKUAN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17625&json=true","fetch_graph":"https://pith.science/api/pith-number/YXZWKUANA4D4R6DRXE4DEGWT5C/graph.json","fetch_events":"https://pith.science/api/pith-number/YXZWKUANA4D4R6DRXE4DEGWT5C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C/action/storage_attestation","attest_author":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C/action/author_attestation","sign_citation":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C/action/citation_signature","submit_replication":"https://pith.science/pith/YXZWKUANA4D4R6DRXE4DEGWT5C/action/replication_record"}},"created_at":"2026-07-05T10:19:41.375140+00:00","updated_at":"2026-07-05T10:19:41.375140+00:00"}