{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3EFZODKPF7WQCZ665VXCXW2GQB","short_pith_number":"pith:3EFZODKP","schema_version":"1.0","canonical_sha256":"d90b970d4f2fed0167deed6e2bdb4680501b6f24299b3ec25844ca345f61f461","source":{"kind":"arxiv","id":"2402.10289","version":1},"attestation_state":"computed","paper":{"title":"Thompson Sampling in Partially Observable Contextual Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Hongju Park, Mohamad Kazem Shirani Faradonbeh","submitted_at":"2024-02-15T19:37:39Z","abstract_excerpt":"Contextual bandits constitute a classical framework for decision-making under uncertainty. In this setting, the goal is to learn the arms of highest reward subject to contextual information, while the unknown reward parameters of each arm need to be learned by experimenting that specific arm. Accordingly, a fundamental problem is that of balancing exploration (i.e., pulling different arms to learn their parameters), versus exploitation (i.e., pulling the best arms to gain reward). To study this problem, the existing literature mostly considers perfectly observed contexts. However, the setting "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10289","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2024-02-15T19:37:39Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3fd04ec65ae92b4378a834b952b3857c52902d7a1311b081afc08735706e3042","abstract_canon_sha256":"a6bec3e7262830d06ba76623d6a6bcfa96c27f6eaba438862645ee5147d95021"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:45:54.416478Z","signature_b64":"GJKP3w8MHJswIVwGJvPUXxUz/7yGFO1KKtyxrYfRfhPgtD6stSmo46raR6Sqoo8HiKi1iPi0YH6IP/HuAsjJDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d90b970d4f2fed0167deed6e2bdb4680501b6f24299b3ec25844ca345f61f461","last_reissued_at":"2026-07-05T07:45:54.415979Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:45:54.415979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Thompson Sampling in Partially Observable Contextual Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Hongju Park, Mohamad Kazem Shirani Faradonbeh","submitted_at":"2024-02-15T19:37:39Z","abstract_excerpt":"Contextual bandits constitute a classical framework for decision-making under uncertainty. In this setting, the goal is to learn the arms of highest reward subject to contextual information, while the unknown reward parameters of each arm need to be learned by experimenting that specific arm. Accordingly, a fundamental problem is that of balancing exploration (i.e., pulling different arms to learn their parameters), versus exploitation (i.e., pulling the best arms to gain reward). To study this problem, the existing literature mostly considers perfectly observed contexts. However, the setting "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10289","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10289/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10289","created_at":"2026-07-05T07:45:54.416040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10289v1","created_at":"2026-07-05T07:45:54.416040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10289","created_at":"2026-07-05T07:45:54.416040+00:00"},{"alias_kind":"pith_short_12","alias_value":"3EFZODKPF7WQ","created_at":"2026-07-05T07:45:54.416040+00:00"},{"alias_kind":"pith_short_16","alias_value":"3EFZODKPF7WQCZ66","created_at":"2026-07-05T07:45:54.416040+00:00"},{"alias_kind":"pith_short_8","alias_value":"3EFZODKP","created_at":"2026-07-05T07:45:54.416040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.16918","citing_title":"Scalable and Interpretable Contextual Bandits: A Literature Review and Retail Offer Prototype","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB","json":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB.json","graph_json":"https://pith.science/api/pith-number/3EFZODKPF7WQCZ665VXCXW2GQB/graph.json","events_json":"https://pith.science/api/pith-number/3EFZODKPF7WQCZ665VXCXW2GQB/events.json","paper":"https://pith.science/paper/3EFZODKP"},"agent_actions":{"view_html":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB","download_json":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB.json","view_paper":"https://pith.science/paper/3EFZODKP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10289&json=true","fetch_graph":"https://pith.science/api/pith-number/3EFZODKPF7WQCZ665VXCXW2GQB/graph.json","fetch_events":"https://pith.science/api/pith-number/3EFZODKPF7WQCZ665VXCXW2GQB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB/action/storage_attestation","attest_author":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB/action/author_attestation","sign_citation":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB/action/citation_signature","submit_replication":"https://pith.science/pith/3EFZODKPF7WQCZ665VXCXW2GQB/action/replication_record"}},"created_at":"2026-07-05T07:45:54.416040+00:00","updated_at":"2026-07-05T07:45:54.416040+00:00"}