{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UEKCKSUD6FAJ7RAVZLLW6C4GJD","short_pith_number":"pith:UEKCKSUD","schema_version":"1.0","canonical_sha256":"a114254a83f1409fc415cad76f0b8648fcfcb4ff8963e43c9c55c5142cb0969d","source":{"kind":"arxiv","id":"2504.06820","version":2},"attestation_state":"computed","paper":{"title":"Regret Bounds for Robust Online Decision Making","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander Appel, Vanessa Kosoy","submitted_at":"2025-04-09T12:25:00Z","abstract_excerpt":"We propose a framework which generalizes \"decision making with structured observations\" by allowing robust (i.e. multivalued) models. In this framework, each model associates each decision with a convex set of probability distributions over outcomes. Nature can choose distributions out of this set in an arbitrary (adversarial) manner, that can be nonoblivious and depend on past history. The resulting framework offers much greater generality than classical bandits and reinforcement learning, since the realizability assumption becomes much weaker and more realistic. We then derive a theory of re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.06820","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-09T12:25:00Z","cross_cats_sorted":[],"title_canon_sha256":"a498b082a19ee76d7e42ffc00136b12f00514e2b0edc67b5115d8128fd9a6d4e","abstract_canon_sha256":"42e72ab65fcbb0336e9c2f9221e3cd9a47c7d289596eea98fd39fe680b1c6774"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:28.982972Z","signature_b64":"E1CM6rf55rIYknC8TgCj4ppybJWpS805XSUHChUAtx2RgZo5P6t1HD/H7CU4EBRzEtAIYjlP2qFQoFPL5bimAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a114254a83f1409fc415cad76f0b8648fcfcb4ff8963e43c9c55c5142cb0969d","last_reissued_at":"2026-07-05T11:27:28.982483Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:28.982483Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Regret Bounds for Robust Online Decision Making","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander Appel, Vanessa Kosoy","submitted_at":"2025-04-09T12:25:00Z","abstract_excerpt":"We propose a framework which generalizes \"decision making with structured observations\" by allowing robust (i.e. multivalued) models. In this framework, each model associates each decision with a convex set of probability distributions over outcomes. Nature can choose distributions out of this set in an arbitrary (adversarial) manner, that can be nonoblivious and depend on past history. The resulting framework offers much greater generality than classical bandits and reinforcement learning, since the realizability assumption becomes much weaker and more realistic. We then derive a theory of re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.06820","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.06820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.06820","created_at":"2026-07-05T11:27:28.982541+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.06820v2","created_at":"2026-07-05T11:27:28.982541+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.06820","created_at":"2026-07-05T11:27:28.982541+00:00"},{"alias_kind":"pith_short_12","alias_value":"UEKCKSUD6FAJ","created_at":"2026-07-05T11:27:28.982541+00:00"},{"alias_kind":"pith_short_16","alias_value":"UEKCKSUD6FAJ7RAV","created_at":"2026-07-05T11:27:28.982541+00:00"},{"alias_kind":"pith_short_8","alias_value":"UEKCKSUD","created_at":"2026-07-05T11:27:28.982541+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23146","citing_title":"Infra-Bayesian Reinforcement Learning Agents Outperform Classical RL For Worst-Case Robustness","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD","json":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD.json","graph_json":"https://pith.science/api/pith-number/UEKCKSUD6FAJ7RAVZLLW6C4GJD/graph.json","events_json":"https://pith.science/api/pith-number/UEKCKSUD6FAJ7RAVZLLW6C4GJD/events.json","paper":"https://pith.science/paper/UEKCKSUD"},"agent_actions":{"view_html":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD","download_json":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD.json","view_paper":"https://pith.science/paper/UEKCKSUD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.06820&json=true","fetch_graph":"https://pith.science/api/pith-number/UEKCKSUD6FAJ7RAVZLLW6C4GJD/graph.json","fetch_events":"https://pith.science/api/pith-number/UEKCKSUD6FAJ7RAVZLLW6C4GJD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD/action/storage_attestation","attest_author":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD/action/author_attestation","sign_citation":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD/action/citation_signature","submit_replication":"https://pith.science/pith/UEKCKSUD6FAJ7RAVZLLW6C4GJD/action/replication_record"}},"created_at":"2026-07-05T11:27:28.982541+00:00","updated_at":"2026-07-05T11:27:28.982541+00:00"}