{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NFCGWMSPPY7SKCE6LES52E7DMT","short_pith_number":"pith:NFCGWMSP","schema_version":"1.0","canonical_sha256":"69446b324f7e3f25089e5925dd13e364fc884fe0551b01fd2ba1b5610ed72abe","source":{"kind":"arxiv","id":"2306.07465","version":2},"attestation_state":"computed","paper":{"title":"A Black-box Approach for Non-stationary Multi-agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Haozhe Jiang, Maryam Fazel, Qiwen Cui, Simon S. Du, Zhihan Xiong","submitted_at":"2023-06-12T23:48:24Z","abstract_excerpt":"We investigate learning the equilibria in non-stationary multi-agent systems and address the challenges that differentiate multi-agent learning from single-agent learning. Specifically, we focus on games with bandit feedback, where testing an equilibrium can result in substantial regret even when the gap to be tested is small, and the existence of multiple optimal solutions (equilibria) in stationary games poses extra challenges. To overcome these obstacles, we propose a versatile black-box approach applicable to a broad spectrum of problems, such as general-sum games, potential games, and Mar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.07465","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-12T23:48:24Z","cross_cats_sorted":["cs.AI","cs.GT","cs.MA","stat.ML"],"title_canon_sha256":"c0f0154f60b2a9066afc3705a0ab559011f7f0a748509bcb5bff04ede54b542e","abstract_canon_sha256":"f151de9cdea359a4f0c6f49e76ca29dce946e31cc6baf1152957f026fc6bd2e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:14:48.874287Z","signature_b64":"CYChrEc1Oa04KS417ZCO1SBNn+kDOTNd7+JxqczxR+0qWUVTrc3STVWjOppt5g3Tg7KwxpOvNjGNFO3HQVm4Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69446b324f7e3f25089e5925dd13e364fc884fe0551b01fd2ba1b5610ed72abe","last_reissued_at":"2026-07-05T08:14:48.873810Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:14:48.873810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Black-box Approach for Non-stationary Multi-agent Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT","cs.MA","stat.ML"],"primary_cat":"cs.LG","authors_text":"Haozhe Jiang, Maryam Fazel, Qiwen Cui, Simon S. Du, Zhihan Xiong","submitted_at":"2023-06-12T23:48:24Z","abstract_excerpt":"We investigate learning the equilibria in non-stationary multi-agent systems and address the challenges that differentiate multi-agent learning from single-agent learning. Specifically, we focus on games with bandit feedback, where testing an equilibrium can result in substantial regret even when the gap to be tested is small, and the existence of multiple optimal solutions (equilibria) in stationary games poses extra challenges. To overcome these obstacles, we propose a versatile black-box approach applicable to a broad spectrum of problems, such as general-sum games, potential games, and Mar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.07465","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.07465/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.07465","created_at":"2026-07-05T08:14:48.873868+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.07465v2","created_at":"2026-07-05T08:14:48.873868+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.07465","created_at":"2026-07-05T08:14:48.873868+00:00"},{"alias_kind":"pith_short_12","alias_value":"NFCGWMSPPY7S","created_at":"2026-07-05T08:14:48.873868+00:00"},{"alias_kind":"pith_short_16","alias_value":"NFCGWMSPPY7SKCE6","created_at":"2026-07-05T08:14:48.873868+00:00"},{"alias_kind":"pith_short_8","alias_value":"NFCGWMSP","created_at":"2026-07-05T08:14:48.873868+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT","json":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT.json","graph_json":"https://pith.science/api/pith-number/NFCGWMSPPY7SKCE6LES52E7DMT/graph.json","events_json":"https://pith.science/api/pith-number/NFCGWMSPPY7SKCE6LES52E7DMT/events.json","paper":"https://pith.science/paper/NFCGWMSP"},"agent_actions":{"view_html":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT","download_json":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT.json","view_paper":"https://pith.science/paper/NFCGWMSP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.07465&json=true","fetch_graph":"https://pith.science/api/pith-number/NFCGWMSPPY7SKCE6LES52E7DMT/graph.json","fetch_events":"https://pith.science/api/pith-number/NFCGWMSPPY7SKCE6LES52E7DMT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT/action/storage_attestation","attest_author":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT/action/author_attestation","sign_citation":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT/action/citation_signature","submit_replication":"https://pith.science/pith/NFCGWMSPPY7SKCE6LES52E7DMT/action/replication_record"}},"created_at":"2026-07-05T08:14:48.873868+00:00","updated_at":"2026-07-05T08:14:48.873868+00:00"}