{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7NWWFO5N4OENKLOSKQHLULDLAG","short_pith_number":"pith:7NWWFO5N","schema_version":"1.0","canonical_sha256":"fb6d62bbade388d52dd2540eba2c6b01ae0085528ecc6ffc11ef120be1c1b391","source":{"kind":"arxiv","id":"2310.00968","version":2},"attestation_state":"computed","paper":{"title":"Variance-Aware Regret Bounds for Stochastic Contextual Dueling Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Farzad Farnoud, Heyang Zhao, Qiwei Di, Quanquan Gu, Tao Jin, Yue Wu","submitted_at":"2023-10-02T08:15:52Z","abstract_excerpt":"Dueling bandits is a prominent framework for decision-making involving preferential feedback, a valuable feature that fits various applications involving human interaction, such as ranking, information retrieval, and recommendation systems. While substantial efforts have been made to minimize the cumulative regret in dueling bandits, a notable gap in the current research is the absence of regret bounds that account for the inherent uncertainty in pairwise comparisons between the dueling arms. Intuitively, greater uncertainty suggests a higher level of difficulty in the problem. To bridge this "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.00968","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-02T08:15:52Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"16f3bab514e48306e80a406f49a4447f22b985d40538c2af19cb17289e8acd6f","abstract_canon_sha256":"86ab6be3994507aa0bbc80f5e3912d6bb2de75c5ae27e8f4063b16324bf789a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:37.788586Z","signature_b64":"IW/Mek462rHJEPh5fWYdhVHYX1xrAXKC7YjmVUsFLyXBjp4lYNFf98aruoHnSVk8HLQGyabQjF7QOPwVe5cmAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb6d62bbade388d52dd2540eba2c6b01ae0085528ecc6ffc11ef120be1c1b391","last_reissued_at":"2026-07-05T09:20:37.788090Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:37.788090Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Variance-Aware Regret Bounds for Stochastic Contextual Dueling Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Farzad Farnoud, Heyang Zhao, Qiwei Di, Quanquan Gu, Tao Jin, Yue Wu","submitted_at":"2023-10-02T08:15:52Z","abstract_excerpt":"Dueling bandits is a prominent framework for decision-making involving preferential feedback, a valuable feature that fits various applications involving human interaction, such as ranking, information retrieval, and recommendation systems. While substantial efforts have been made to minimize the cumulative regret in dueling bandits, a notable gap in the current research is the absence of regret bounds that account for the inherent uncertainty in pairwise comparisons between the dueling arms. Intuitively, greater uncertainty suggests a higher level of difficulty in the problem. To bridge this "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00968","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.00968","created_at":"2026-07-05T09:20:37.788147+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.00968v2","created_at":"2026-07-05T09:20:37.788147+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00968","created_at":"2026-07-05T09:20:37.788147+00:00"},{"alias_kind":"pith_short_12","alias_value":"7NWWFO5N4OEN","created_at":"2026-07-05T09:20:37.788147+00:00"},{"alias_kind":"pith_short_16","alias_value":"7NWWFO5N4OENKLOS","created_at":"2026-07-05T09:20:37.788147+00:00"},{"alias_kind":"pith_short_8","alias_value":"7NWWFO5N","created_at":"2026-07-05T09:20:37.788147+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.19241","citing_title":"ActiveDPO: Active Direct Preference Optimization for Sample-Efficient Alignment","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22161","citing_title":"Logistic Bandits with $\\tilde{O}(\\sqrt{dT})$ Regret without Context Diversity Assumptions","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG","json":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG.json","graph_json":"https://pith.science/api/pith-number/7NWWFO5N4OENKLOSKQHLULDLAG/graph.json","events_json":"https://pith.science/api/pith-number/7NWWFO5N4OENKLOSKQHLULDLAG/events.json","paper":"https://pith.science/paper/7NWWFO5N"},"agent_actions":{"view_html":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG","download_json":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG.json","view_paper":"https://pith.science/paper/7NWWFO5N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.00968&json=true","fetch_graph":"https://pith.science/api/pith-number/7NWWFO5N4OENKLOSKQHLULDLAG/graph.json","fetch_events":"https://pith.science/api/pith-number/7NWWFO5N4OENKLOSKQHLULDLAG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG/action/storage_attestation","attest_author":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG/action/author_attestation","sign_citation":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG/action/citation_signature","submit_replication":"https://pith.science/pith/7NWWFO5N4OENKLOSKQHLULDLAG/action/replication_record"}},"created_at":"2026-07-05T09:20:37.788147+00:00","updated_at":"2026-07-05T09:20:37.788147+00:00"}