{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6AQ4ZACJENPW2K2NTBGAQGJ7CL","short_pith_number":"pith:6AQ4ZACJ","schema_version":"1.0","canonical_sha256":"f021cc8049235f6d2b4d984c08193f12e588a134b59b9b4b5d29efa4d4b37e82","source":{"kind":"arxiv","id":"2305.16074","version":1},"attestation_state":"computed","paper":{"title":"Combinatorial Bandits for Maximum Value Reward Function under Max Value-Index Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.ST","stat.TH"],"primary_cat":"cs.LG","authors_text":"Milan Vojnovi\\'c, Wei Chen, Yiliu Wang","submitted_at":"2023-05-25T14:02:12Z","abstract_excerpt":"We consider a combinatorial multi-armed bandit problem for maximum value reward function under maximum value and index feedback. This is a new feedback structure that lies in between commonly studied semi-bandit and full-bandit feedback structures. We propose an algorithm and provide a regret bound for problem instances with stochastic arm outcomes according to arbitrary distributions with finite supports. The regret analysis rests on considering an extended set of arms, associated with values and probabilities of arm outcomes, and applying a smoothness condition. Our algorithm achieves a $O(("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16074","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-25T14:02:12Z","cross_cats_sorted":["math.ST","stat.TH"],"title_canon_sha256":"28ac82dc3a6f2d6e1123a28fff4a5afd8b359009d9ab5e25cbea94ffc32b9808","abstract_canon_sha256":"8fe3f4359caf593250d6e8fd757a3212bbc65fb084be80e20d32970f08d29210"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:13:59.241628Z","signature_b64":"h0xhXX5LGbD2OO7VW75KyK3zgOk+vConMcFOtvc6qjRq2epkZvK30eDKSOGt1zBE7iN+o8GTVxhppoHuGu7xDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f021cc8049235f6d2b4d984c08193f12e588a134b59b9b4b5d29efa4d4b37e82","last_reissued_at":"2026-07-05T06:13:59.241266Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:13:59.241266Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Combinatorial Bandits for Maximum Value Reward Function under Max Value-Index Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.ST","stat.TH"],"primary_cat":"cs.LG","authors_text":"Milan Vojnovi\\'c, Wei Chen, Yiliu Wang","submitted_at":"2023-05-25T14:02:12Z","abstract_excerpt":"We consider a combinatorial multi-armed bandit problem for maximum value reward function under maximum value and index feedback. This is a new feedback structure that lies in between commonly studied semi-bandit and full-bandit feedback structures. We propose an algorithm and provide a regret bound for problem instances with stochastic arm outcomes according to arbitrary distributions with finite supports. The regret analysis rests on considering an extended set of arms, associated with values and probabilities of arm outcomes, and applying a smoothness condition. Our algorithm achieves a $O(("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16074","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16074/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16074","created_at":"2026-07-05T06:13:59.241338+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16074v1","created_at":"2026-07-05T06:13:59.241338+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16074","created_at":"2026-07-05T06:13:59.241338+00:00"},{"alias_kind":"pith_short_12","alias_value":"6AQ4ZACJENPW","created_at":"2026-07-05T06:13:59.241338+00:00"},{"alias_kind":"pith_short_16","alias_value":"6AQ4ZACJENPW2K2N","created_at":"2026-07-05T06:13:59.241338+00:00"},{"alias_kind":"pith_short_8","alias_value":"6AQ4ZACJ","created_at":"2026-07-05T06:13:59.241338+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.19300","citing_title":"Offline Learning for Combinatorial Multi-armed Bandits","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL","json":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL.json","graph_json":"https://pith.science/api/pith-number/6AQ4ZACJENPW2K2NTBGAQGJ7CL/graph.json","events_json":"https://pith.science/api/pith-number/6AQ4ZACJENPW2K2NTBGAQGJ7CL/events.json","paper":"https://pith.science/paper/6AQ4ZACJ"},"agent_actions":{"view_html":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL","download_json":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL.json","view_paper":"https://pith.science/paper/6AQ4ZACJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16074&json=true","fetch_graph":"https://pith.science/api/pith-number/6AQ4ZACJENPW2K2NTBGAQGJ7CL/graph.json","fetch_events":"https://pith.science/api/pith-number/6AQ4ZACJENPW2K2NTBGAQGJ7CL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL/action/storage_attestation","attest_author":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL/action/author_attestation","sign_citation":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL/action/citation_signature","submit_replication":"https://pith.science/pith/6AQ4ZACJENPW2K2NTBGAQGJ7CL/action/replication_record"}},"created_at":"2026-07-05T06:13:59.241338+00:00","updated_at":"2026-07-05T06:13:59.241338+00:00"}