{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F2OXQDFXIF6KDMEIQF3DHJJNTR","short_pith_number":"pith:F2OXQDFX","schema_version":"1.0","canonical_sha256":"2e9d780cb7417ca1b088817633a52d9c40d352985c272a79584267a62e31fd68","source":{"kind":"arxiv","id":"2412.18297","version":2},"attestation_state":"computed","paper":{"title":"Learning to Play Against Unknown Opponents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.GT","authors_text":"Eshwar Ram Arunachaleswaran, Jon Schneider, Natalie Collina","submitted_at":"2024-12-24T09:05:06Z","abstract_excerpt":"We consider the problem of a learning agent who has to repeatedly play a general sum game against a strategic opponent who acts to maximize their own payoff by optimally responding against the learner's algorithm. The learning agent knows their own payoff function, but is uncertain about the payoff of their opponent (knowing only that it is drawn from some distribution $\\mathcal{D}$). What learning algorithm should the agent run in order to maximize their own total utility, either in expectation or in the worst-case over $\\mathcal{D}$?\n  When the learning algorithm is constrained to be a no-re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18297","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.GT","submitted_at":"2024-12-24T09:05:06Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"73d797b2a30af09e48086bfcb57d54b541ce495aa3845ca5aade379812fd49ad","abstract_canon_sha256":"23032200cd50ab0bf2e20dca2e50a96d84ee273162282c3fefd52f2645f0b021"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:14.918120Z","signature_b64":"kWg/tvIAUxcqWOsMJGbOUgDfd10pxhTUY+rizqFIoiAlEQYOfiiZ4s0aWw5qYXqFrQTOroYpMbgbJm+O57F6Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e9d780cb7417ca1b088817633a52d9c40d352985c272a79584267a62e31fd68","last_reissued_at":"2026-07-05T10:17:14.917653Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:14.917653Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Play Against Unknown Opponents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.GT","authors_text":"Eshwar Ram Arunachaleswaran, Jon Schneider, Natalie Collina","submitted_at":"2024-12-24T09:05:06Z","abstract_excerpt":"We consider the problem of a learning agent who has to repeatedly play a general sum game against a strategic opponent who acts to maximize their own payoff by optimally responding against the learner's algorithm. The learning agent knows their own payoff function, but is uncertain about the payoff of their opponent (knowing only that it is drawn from some distribution $\\mathcal{D}$). What learning algorithm should the agent run in order to maximize their own total utility, either in expectation or in the worst-case over $\\mathcal{D}$?\n  When the learning algorithm is constrained to be a no-re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18297","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18297/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18297","created_at":"2026-07-05T10:17:14.917711+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18297v2","created_at":"2026-07-05T10:17:14.917711+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18297","created_at":"2026-07-05T10:17:14.917711+00:00"},{"alias_kind":"pith_short_12","alias_value":"F2OXQDFXIF6K","created_at":"2026-07-05T10:17:14.917711+00:00"},{"alias_kind":"pith_short_16","alias_value":"F2OXQDFXIF6KDMEI","created_at":"2026-07-05T10:17:14.917711+00:00"},{"alias_kind":"pith_short_8","alias_value":"F2OXQDFX","created_at":"2026-07-05T10:17:14.917711+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.08597","citing_title":"Markets with Heterogeneous Agents: Dynamics and Survival of Bayesian vs. No-Regret Learners","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR","json":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR.json","graph_json":"https://pith.science/api/pith-number/F2OXQDFXIF6KDMEIQF3DHJJNTR/graph.json","events_json":"https://pith.science/api/pith-number/F2OXQDFXIF6KDMEIQF3DHJJNTR/events.json","paper":"https://pith.science/paper/F2OXQDFX"},"agent_actions":{"view_html":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR","download_json":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR.json","view_paper":"https://pith.science/paper/F2OXQDFX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18297&json=true","fetch_graph":"https://pith.science/api/pith-number/F2OXQDFXIF6KDMEIQF3DHJJNTR/graph.json","fetch_events":"https://pith.science/api/pith-number/F2OXQDFXIF6KDMEIQF3DHJJNTR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR/action/storage_attestation","attest_author":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR/action/author_attestation","sign_citation":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR/action/citation_signature","submit_replication":"https://pith.science/pith/F2OXQDFXIF6KDMEIQF3DHJJNTR/action/replication_record"}},"created_at":"2026-07-05T10:17:14.917711+00:00","updated_at":"2026-07-05T10:17:14.917711+00:00"}