{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:R6AOCAIGC3P3HYVFM7QTFDM4GL","short_pith_number":"pith:R6AOCAIG","schema_version":"1.0","canonical_sha256":"8f80e1010616dfb3e2a567e1328d9c32d842bb8b065cfbae4d7805066262e6e0","source":{"kind":"arxiv","id":"2203.04192","version":2},"attestation_state":"computed","paper":{"title":"Reward-Biased Maximum Likelihood Estimation for Neural Contextual Bandits","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ping-Chun Hsieh, Yu-Heng Hung","submitted_at":"2022-03-08T16:33:36Z","abstract_excerpt":"Reward-biased maximum likelihood estimation (RBMLE) is a classic principle in the adaptive control literature for tackling explore-exploit trade-offs. This paper studies the stochastic contextual bandit problem with general bounded reward functions and proposes NeuralRBMLE, which adapts the RBMLE principle by adding a bias term to the log-likelihood to enforce exploration. NeuralRBMLE leverages the representation power of neural networks and directly encodes exploratory behavior in the parameter space, without constructing confidence intervals of the estimated rewards. We propose two variants "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.04192","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-03-08T16:33:36Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"5a182d5885870c24128e88de98eb95621cb7888c61469fcdbec256b4045350ee","abstract_canon_sha256":"696b259fecf533b0d6bf7e4000685127dd005f1b39e96fa41c4a3bd1d7648d88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:27:11.892725Z","signature_b64":"AbysIhT6RWcoXjapRkw9eSul+8C8/qgRamTZlXuNixoCwCnuwjNHaH2hdSthqYwPNtVnWRLFHzZ9y0ceSmeMDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f80e1010616dfb3e2a567e1328d9c32d842bb8b065cfbae4d7805066262e6e0","last_reissued_at":"2026-07-05T04:27:11.892329Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:27:11.892329Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward-Biased Maximum Likelihood Estimation for Neural Contextual Bandits","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ping-Chun Hsieh, Yu-Heng Hung","submitted_at":"2022-03-08T16:33:36Z","abstract_excerpt":"Reward-biased maximum likelihood estimation (RBMLE) is a classic principle in the adaptive control literature for tackling explore-exploit trade-offs. This paper studies the stochastic contextual bandit problem with general bounded reward functions and proposes NeuralRBMLE, which adapts the RBMLE principle by adding a bias term to the log-likelihood to enforce exploration. NeuralRBMLE leverages the representation power of neural networks and directly encodes exploratory behavior in the parameter space, without constructing confidence intervals of the estimated rewards. We propose two variants "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.04192","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.04192/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.04192","created_at":"2026-07-05T04:27:11.892385+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.04192v2","created_at":"2026-07-05T04:27:11.892385+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.04192","created_at":"2026-07-05T04:27:11.892385+00:00"},{"alias_kind":"pith_short_12","alias_value":"R6AOCAIGC3P3","created_at":"2026-07-05T04:27:11.892385+00:00"},{"alias_kind":"pith_short_16","alias_value":"R6AOCAIGC3P3HYVF","created_at":"2026-07-05T04:27:11.892385+00:00"},{"alias_kind":"pith_short_8","alias_value":"R6AOCAIG","created_at":"2026-07-05T04:27:11.892385+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL","json":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL.json","graph_json":"https://pith.science/api/pith-number/R6AOCAIGC3P3HYVFM7QTFDM4GL/graph.json","events_json":"https://pith.science/api/pith-number/R6AOCAIGC3P3HYVFM7QTFDM4GL/events.json","paper":"https://pith.science/paper/R6AOCAIG"},"agent_actions":{"view_html":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL","download_json":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL.json","view_paper":"https://pith.science/paper/R6AOCAIG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.04192&json=true","fetch_graph":"https://pith.science/api/pith-number/R6AOCAIGC3P3HYVFM7QTFDM4GL/graph.json","fetch_events":"https://pith.science/api/pith-number/R6AOCAIGC3P3HYVFM7QTFDM4GL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL/action/storage_attestation","attest_author":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL/action/author_attestation","sign_citation":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL/action/citation_signature","submit_replication":"https://pith.science/pith/R6AOCAIGC3P3HYVFM7QTFDM4GL/action/replication_record"}},"created_at":"2026-07-05T04:27:11.892385+00:00","updated_at":"2026-07-05T04:27:11.892385+00:00"}