{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ROLQJQ6XUO2CKKKE2RUJGLPFBG","short_pith_number":"pith:ROLQJQ6X","schema_version":"1.0","canonical_sha256":"8b9704c3d7a3b4252944d468932de5098d4b0c768bb9792d4e6d2b5a9931b8eb","source":{"kind":"arxiv","id":"2509.10520","version":1},"attestation_state":"computed","paper":{"title":"Offline Contextual Bandit with Counterfactual Sample Identification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexandre Gilotte, Benjamin Heymann, Imad Aouali, Otmane Sakhi","submitted_at":"2025-09-03T17:23:32Z","abstract_excerpt":"In production systems, contextual bandit approaches often rely on direct reward models that take both action and context as input. However, these models can suffer from confounding, making it difficult to isolate the effect of the action from that of the context. We present \\emph{Counterfactual Sample Identification}, a new approach that re-frames the problem: rather than predicting reward, it learns to recognize which action led to a successful (binary) outcome by comparing it to a counterfactual action sampled from the logging policy under the same context. The method is theoretically ground"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.10520","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-03T17:23:32Z","cross_cats_sorted":[],"title_canon_sha256":"ea081b9de2f1554aa55c83f48599d08d932fe9cedf77084c96d3a3c887931f03","abstract_canon_sha256":"93a34fc5dd19b130c4fbad9bf8c2efff7e6f2005be55afb4bb480119cf66d870"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:22.520555Z","signature_b64":"yOrEv4SjQ8dAus0pUaXTARK84RGMQo/z4JfTO64eCLZaO67hC5xP/LJL1mWtQf4RXg/x7sktTZrCxaALrJUIBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b9704c3d7a3b4252944d468932de5098d4b0c768bb9792d4e6d2b5a9931b8eb","last_reissued_at":"2026-07-05T12:11:22.520088Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:22.520088Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Contextual Bandit with Counterfactual Sample Identification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexandre Gilotte, Benjamin Heymann, Imad Aouali, Otmane Sakhi","submitted_at":"2025-09-03T17:23:32Z","abstract_excerpt":"In production systems, contextual bandit approaches often rely on direct reward models that take both action and context as input. However, these models can suffer from confounding, making it difficult to isolate the effect of the action from that of the context. We present \\emph{Counterfactual Sample Identification}, a new approach that re-frames the problem: rather than predicting reward, it learns to recognize which action led to a successful (binary) outcome by comparing it to a counterfactual action sampled from the logging policy under the same context. The method is theoretically ground"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.10520","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.10520/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.10520","created_at":"2026-07-05T12:11:22.520148+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.10520v1","created_at":"2026-07-05T12:11:22.520148+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.10520","created_at":"2026-07-05T12:11:22.520148+00:00"},{"alias_kind":"pith_short_12","alias_value":"ROLQJQ6XUO2C","created_at":"2026-07-05T12:11:22.520148+00:00"},{"alias_kind":"pith_short_16","alias_value":"ROLQJQ6XUO2CKKKE","created_at":"2026-07-05T12:11:22.520148+00:00"},{"alias_kind":"pith_short_8","alias_value":"ROLQJQ6X","created_at":"2026-07-05T12:11:22.520148+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG","json":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG.json","graph_json":"https://pith.science/api/pith-number/ROLQJQ6XUO2CKKKE2RUJGLPFBG/graph.json","events_json":"https://pith.science/api/pith-number/ROLQJQ6XUO2CKKKE2RUJGLPFBG/events.json","paper":"https://pith.science/paper/ROLQJQ6X"},"agent_actions":{"view_html":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG","download_json":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG.json","view_paper":"https://pith.science/paper/ROLQJQ6X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.10520&json=true","fetch_graph":"https://pith.science/api/pith-number/ROLQJQ6XUO2CKKKE2RUJGLPFBG/graph.json","fetch_events":"https://pith.science/api/pith-number/ROLQJQ6XUO2CKKKE2RUJGLPFBG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG/action/storage_attestation","attest_author":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG/action/author_attestation","sign_citation":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG/action/citation_signature","submit_replication":"https://pith.science/pith/ROLQJQ6XUO2CKKKE2RUJGLPFBG/action/replication_record"}},"created_at":"2026-07-05T12:11:22.520148+00:00","updated_at":"2026-07-05T12:11:22.520148+00:00"}