{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5TGGHEPWKIJRI44NOK2WFGLYYC","short_pith_number":"pith:5TGGHEPW","schema_version":"1.0","canonical_sha256":"eccc6391f6521314738d72b5629978c084c13836f05cc7add5f2e70ae87cb5bc","source":{"kind":"arxiv","id":"2502.01118","version":1},"attestation_state":"computed","paper":{"title":"Large Language Model-Enhanced Multi-Armed Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chenjun Xiao, Jiahang Sun, John C.S. Lui, Runhan Yang, Zhiyong Wang, Zhongxiang Dai","submitted_at":"2025-02-03T07:19:05Z","abstract_excerpt":"Large language models (LLMs) have been adopted to solve sequential decision-making tasks such as multi-armed bandits (MAB), in which an LLM is directly instructed to select the arms to pull in every iteration. However, this paradigm of direct arm selection using LLMs has been shown to be suboptimal in many MAB tasks. Therefore, we propose an alternative approach which combines the strengths of classical MAB and LLMs. Specifically, we adopt a classical MAB algorithm as the high-level framework and leverage the strong in-context learning capability of LLMs to perform the sub-task of reward predi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01118","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T07:19:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b6cb6c17fe416e916992ea50dc372a0cf392dac92b1bd4fc3652125d7c548cd3","abstract_canon_sha256":"37d0818a6e3be02ceb184cb4170456f088dd6f3ee9badd7d8b0799bd4b4ed28f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:43.059078Z","signature_b64":"2GNI1rU3HE1gVlweUxEKZg8N/ynFM5n5jwQZqX2ykIEgsZXUCP4/z8D+X+GNgmZ9Tnv7OFeCcnkp7GQU82YSBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eccc6391f6521314738d72b5629978c084c13836f05cc7add5f2e70ae87cb5bc","last_reissued_at":"2026-07-05T10:08:43.058599Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:43.058599Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Model-Enhanced Multi-Armed Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chenjun Xiao, Jiahang Sun, John C.S. Lui, Runhan Yang, Zhiyong Wang, Zhongxiang Dai","submitted_at":"2025-02-03T07:19:05Z","abstract_excerpt":"Large language models (LLMs) have been adopted to solve sequential decision-making tasks such as multi-armed bandits (MAB), in which an LLM is directly instructed to select the arms to pull in every iteration. However, this paradigm of direct arm selection using LLMs has been shown to be suboptimal in many MAB tasks. Therefore, we propose an alternative approach which combines the strengths of classical MAB and LLMs. Specifically, we adopt a classical MAB algorithm as the high-level framework and leverage the strong in-context learning capability of LLMs to perform the sub-task of reward predi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01118","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01118/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01118","created_at":"2026-07-05T10:08:43.058657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01118v1","created_at":"2026-07-05T10:08:43.058657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01118","created_at":"2026-07-05T10:08:43.058657+00:00"},{"alias_kind":"pith_short_12","alias_value":"5TGGHEPWKIJR","created_at":"2026-07-05T10:08:43.058657+00:00"},{"alias_kind":"pith_short_16","alias_value":"5TGGHEPWKIJRI44N","created_at":"2026-07-05T10:08:43.058657+00:00"},{"alias_kind":"pith_short_8","alias_value":"5TGGHEPW","created_at":"2026-07-05T10:08:43.058657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23299","citing_title":"GRIMIP: A General Framework for Instance-Specific Configuration of MIP Solvers Using LLMs","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05859","citing_title":"When Do We Need LLMs? A Diagnostic for Language-Driven Bandits","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14961","citing_title":"Calibration-Gated LLM Pseudo-Observations for Online Contextual Bandits","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC","json":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC.json","graph_json":"https://pith.science/api/pith-number/5TGGHEPWKIJRI44NOK2WFGLYYC/graph.json","events_json":"https://pith.science/api/pith-number/5TGGHEPWKIJRI44NOK2WFGLYYC/events.json","paper":"https://pith.science/paper/5TGGHEPW"},"agent_actions":{"view_html":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC","download_json":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC.json","view_paper":"https://pith.science/paper/5TGGHEPW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01118&json=true","fetch_graph":"https://pith.science/api/pith-number/5TGGHEPWKIJRI44NOK2WFGLYYC/graph.json","fetch_events":"https://pith.science/api/pith-number/5TGGHEPWKIJRI44NOK2WFGLYYC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC/action/storage_attestation","attest_author":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC/action/author_attestation","sign_citation":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC/action/citation_signature","submit_replication":"https://pith.science/pith/5TGGHEPWKIJRI44NOK2WFGLYYC/action/replication_record"}},"created_at":"2026-07-05T10:08:43.058657+00:00","updated_at":"2026-07-05T10:08:43.058657+00:00"}