{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2EJJCM5SHEK6ZVX6N5DTHKLGW2","short_pith_number":"pith:2EJJCM5S","schema_version":"1.0","canonical_sha256":"d1129133b23915ecd6fe6f4733a966b6b1302d32198768ec8258b55a9ef7990c","source":{"kind":"arxiv","id":"2310.11515","version":1},"attestation_state":"computed","paper":{"title":"Value-Biased Maximum Likelihood Estimation for Model-based Reinforcement Learning in Discounted Linear MDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akshay Mete, Ping-Chun Hsieh, P. R. Kumar, Yu-Heng Hung","submitted_at":"2023-10-17T18:27:27Z","abstract_excerpt":"We consider the infinite-horizon linear Markov Decision Processes (MDPs), where the transition probabilities of the dynamic model can be linearly parameterized with the help of a predefined low-dimensional feature mapping. While the existing regression-based approaches have been theoretically shown to achieve nearly-optimal regret, they are computationally rather inefficient due to the need for a large number of optimization runs in each time step, especially when the state and action spaces are large. To address this issue, we propose to solve linear MDPs through the lens of Value-Biased Maxi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.11515","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-17T18:27:27Z","cross_cats_sorted":[],"title_canon_sha256":"41e6213341c79b98ee038604f8f3689996f58d78b5673f98ec472f912d2f357c","abstract_canon_sha256":"01616890cc305502f05713469299aefebbf2c345ee8ee28fae066615391b518b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:08.451096Z","signature_b64":"Zusa+L0d/a+QksL60LGnJfqE6oPLuo8UXHZMayGmk47X7KQpjAgxtBnDdpM2K0dozE7up6gkG7ZbndhOkXalCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1129133b23915ecd6fe6f4733a966b6b1302d32198768ec8258b55a9ef7990c","last_reissued_at":"2026-07-05T07:02:08.450642Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:08.450642Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Value-Biased Maximum Likelihood Estimation for Model-based Reinforcement Learning in Discounted Linear MDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akshay Mete, Ping-Chun Hsieh, P. R. Kumar, Yu-Heng Hung","submitted_at":"2023-10-17T18:27:27Z","abstract_excerpt":"We consider the infinite-horizon linear Markov Decision Processes (MDPs), where the transition probabilities of the dynamic model can be linearly parameterized with the help of a predefined low-dimensional feature mapping. While the existing regression-based approaches have been theoretically shown to achieve nearly-optimal regret, they are computationally rather inefficient due to the need for a large number of optimization runs in each time step, especially when the state and action spaces are large. To address this issue, we propose to solve linear MDPs through the lens of Value-Biased Maxi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.11515","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.11515/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.11515","created_at":"2026-07-05T07:02:08.450699+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.11515v1","created_at":"2026-07-05T07:02:08.450699+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.11515","created_at":"2026-07-05T07:02:08.450699+00:00"},{"alias_kind":"pith_short_12","alias_value":"2EJJCM5SHEK6","created_at":"2026-07-05T07:02:08.450699+00:00"},{"alias_kind":"pith_short_16","alias_value":"2EJJCM5SHEK6ZVX6","created_at":"2026-07-05T07:02:08.450699+00:00"},{"alias_kind":"pith_short_8","alias_value":"2EJJCM5S","created_at":"2026-07-05T07:02:08.450699+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2","json":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2.json","graph_json":"https://pith.science/api/pith-number/2EJJCM5SHEK6ZVX6N5DTHKLGW2/graph.json","events_json":"https://pith.science/api/pith-number/2EJJCM5SHEK6ZVX6N5DTHKLGW2/events.json","paper":"https://pith.science/paper/2EJJCM5S"},"agent_actions":{"view_html":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2","download_json":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2.json","view_paper":"https://pith.science/paper/2EJJCM5S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.11515&json=true","fetch_graph":"https://pith.science/api/pith-number/2EJJCM5SHEK6ZVX6N5DTHKLGW2/graph.json","fetch_events":"https://pith.science/api/pith-number/2EJJCM5SHEK6ZVX6N5DTHKLGW2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2/action/storage_attestation","attest_author":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2/action/author_attestation","sign_citation":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2/action/citation_signature","submit_replication":"https://pith.science/pith/2EJJCM5SHEK6ZVX6N5DTHKLGW2/action/replication_record"}},"created_at":"2026-07-05T07:02:08.450699+00:00","updated_at":"2026-07-05T07:02:08.450699+00:00"}