{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CVIWYU6QRQ3MJVK7ZTLGK4SELK","short_pith_number":"pith:CVIWYU6Q","schema_version":"1.0","canonical_sha256":"15516c53d08c36c4d55fccd66572445aacae599a68b570bec13b22103c4c96fd","source":{"kind":"arxiv","id":"2311.02268","version":1},"attestation_state":"computed","paper":{"title":"LLMs-augmented Contextual Bandit","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ali Baheri, Cecilia O. Alm","submitted_at":"2023-11-03T23:12:57Z","abstract_excerpt":"Contextual bandits have emerged as a cornerstone in reinforcement learning, enabling systems to make decisions with partial feedback. However, as contexts grow in complexity, traditional bandit algorithms can face challenges in adequately capturing and utilizing such contexts. In this paper, we propose a novel integration of large language models (LLMs) with the contextual bandit framework. By leveraging LLMs as an encoder, we enrich the representation of the context, providing the bandit with a denser and more informative view. Preliminary results on synthetic datasets demonstrate the potenti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.02268","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T23:12:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"787494a13642c97a0fa2fa7cbb1d11e4b889cea634f48feecf59abcd80023940","abstract_canon_sha256":"3dd0019e2b1fd4d3375ec549ad5f67b5d04c954925077581706805613d1180d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:08.579117Z","signature_b64":"476pITNAnFKQaQaui4GnxvsOnOygb3NFken9BCO4+GuUBT52kiSgw79cNZz8U2ZXtct2o97WdUL2y77kkkSxAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15516c53d08c36c4d55fccd66572445aacae599a68b570bec13b22103c4c96fd","last_reissued_at":"2026-07-05T07:09:08.578611Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:08.578611Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs-augmented Contextual Bandit","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ali Baheri, Cecilia O. Alm","submitted_at":"2023-11-03T23:12:57Z","abstract_excerpt":"Contextual bandits have emerged as a cornerstone in reinforcement learning, enabling systems to make decisions with partial feedback. However, as contexts grow in complexity, traditional bandit algorithms can face challenges in adequately capturing and utilizing such contexts. In this paper, we propose a novel integration of large language models (LLMs) with the contextual bandit framework. By leveraging LLMs as an encoder, we enrich the representation of the context, providing the bandit with a denser and more informative view. Preliminary results on synthetic datasets demonstrate the potenti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02268","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.02268/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.02268","created_at":"2026-07-05T07:09:08.578679+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.02268v1","created_at":"2026-07-05T07:09:08.578679+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02268","created_at":"2026-07-05T07:09:08.578679+00:00"},{"alias_kind":"pith_short_12","alias_value":"CVIWYU6QRQ3M","created_at":"2026-07-05T07:09:08.578679+00:00"},{"alias_kind":"pith_short_16","alias_value":"CVIWYU6QRQ3MJVK7","created_at":"2026-07-05T07:09:08.578679+00:00"},{"alias_kind":"pith_short_8","alias_value":"CVIWYU6Q","created_at":"2026-07-05T07:09:08.578679+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23933","citing_title":"Flow-Corrected Thompson Sampling for Non-Stationary Contextual Bandits","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23933","citing_title":"Flow-Corrected Thompson Sampling for Non-Stationary Contextual Bandits","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05859","citing_title":"When Do We Need LLMs? A Diagnostic for Language-Driven Bandits","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05417","citing_title":"Multi-Drafter Speculative Decoding with Alignment Feedback","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14961","citing_title":"Calibration-Gated LLM Pseudo-Observations for Online Contextual Bandits","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK","json":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK.json","graph_json":"https://pith.science/api/pith-number/CVIWYU6QRQ3MJVK7ZTLGK4SELK/graph.json","events_json":"https://pith.science/api/pith-number/CVIWYU6QRQ3MJVK7ZTLGK4SELK/events.json","paper":"https://pith.science/paper/CVIWYU6Q"},"agent_actions":{"view_html":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK","download_json":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK.json","view_paper":"https://pith.science/paper/CVIWYU6Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.02268&json=true","fetch_graph":"https://pith.science/api/pith-number/CVIWYU6QRQ3MJVK7ZTLGK4SELK/graph.json","fetch_events":"https://pith.science/api/pith-number/CVIWYU6QRQ3MJVK7ZTLGK4SELK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK/action/storage_attestation","attest_author":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK/action/author_attestation","sign_citation":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK/action/citation_signature","submit_replication":"https://pith.science/pith/CVIWYU6QRQ3MJVK7ZTLGK4SELK/action/replication_record"}},"created_at":"2026-07-05T07:09:08.578679+00:00","updated_at":"2026-07-05T07:09:08.578679+00:00"}