{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:YCJAW5T2A6JMQTO6U77DKIEPLU","short_pith_number":"pith:YCJAW5T2","schema_version":"1.0","canonical_sha256":"c0920b767a0792c84ddea7fe35208f5d0e3043e90b678c798dbe34ae3be57169","source":{"kind":"arxiv","id":"1907.05388","version":2},"attestation_state":"computed","paper":{"title":"Provably Efficient Reinforcement Learning with Linear Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Michael I. Jordan, Zhaoran Wang, Zhuoran Yang","submitted_at":"2019-07-11T17:06:11Z","abstract_excerpt":"Modern Reinforcement Learning (RL) is commonly applied to practical problems with an enormous number of states, where function approximation must be deployed to approximate either the value function or the policy. The introduction of function approximation raises a fundamental set of challenges involving computational and statistical efficiency, especially given the need to manage the exploration/exploitation tradeoff. As a result, a core RL question remains open: how can we design provably efficient RL algorithms that incorporate function approximation? This question persists even in a basic "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1907.05388","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-07-11T17:06:11Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"4fe9cd94fb6c5ffc4ed4196f1b5e9569309a272ab97b16966557e34aabcf22b9","abstract_canon_sha256":"14f2bf426fd3abdd2d3a2f88476a1361682a98eeac10bdf71b184c2632b6c73c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T23:52:21.973098Z","signature_b64":"Tfy7AIh1qf+Cox5ba+bqQ/g1yYC8FiBRV2zmedzAM0Wq11ARCoIfIXuPYEJprP/sUg3WhUHpTP2IR9p8X4oCBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0920b767a0792c84ddea7fe35208f5d0e3043e90b678c798dbe34ae3be57169","last_reissued_at":"2026-07-04T23:52:21.972674Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T23:52:21.972674Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Provably Efficient Reinforcement Learning with Linear Function Approximation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Michael I. Jordan, Zhaoran Wang, Zhuoran Yang","submitted_at":"2019-07-11T17:06:11Z","abstract_excerpt":"Modern Reinforcement Learning (RL) is commonly applied to practical problems with an enormous number of states, where function approximation must be deployed to approximate either the value function or the policy. The introduction of function approximation raises a fundamental set of challenges involving computational and statistical efficiency, especially given the need to manage the exploration/exploitation tradeoff. As a result, a core RL question remains open: how can we design provably efficient RL algorithms that incorporate function approximation? This question persists even in a basic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1907.05388","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1907.05388/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1907.05388","created_at":"2026-07-04T23:52:21.972741+00:00"},{"alias_kind":"arxiv_version","alias_value":"1907.05388v2","created_at":"2026-07-04T23:52:21.972741+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1907.05388","created_at":"2026-07-04T23:52:21.972741+00:00"},{"alias_kind":"pith_short_12","alias_value":"YCJAW5T2A6JM","created_at":"2026-07-04T23:52:21.972741+00:00"},{"alias_kind":"pith_short_16","alias_value":"YCJAW5T2A6JMQTO6","created_at":"2026-07-04T23:52:21.972741+00:00"},{"alias_kind":"pith_short_8","alias_value":"YCJAW5T2","created_at":"2026-07-04T23:52:21.972741+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11171","citing_title":"Bellman-sufficient Information Complexity","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11171","citing_title":"Bellman-sufficient Information Complexity","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03125","citing_title":"Taming the Curses of Multiagency in Robust Markov Games with Large State Space through Linear Function Approximation","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU","json":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU.json","graph_json":"https://pith.science/api/pith-number/YCJAW5T2A6JMQTO6U77DKIEPLU/graph.json","events_json":"https://pith.science/api/pith-number/YCJAW5T2A6JMQTO6U77DKIEPLU/events.json","paper":"https://pith.science/paper/YCJAW5T2"},"agent_actions":{"view_html":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU","download_json":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU.json","view_paper":"https://pith.science/paper/YCJAW5T2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1907.05388&json=true","fetch_graph":"https://pith.science/api/pith-number/YCJAW5T2A6JMQTO6U77DKIEPLU/graph.json","fetch_events":"https://pith.science/api/pith-number/YCJAW5T2A6JMQTO6U77DKIEPLU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU/action/storage_attestation","attest_author":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU/action/author_attestation","sign_citation":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU/action/citation_signature","submit_replication":"https://pith.science/pith/YCJAW5T2A6JMQTO6U77DKIEPLU/action/replication_record"}},"created_at":"2026-07-04T23:52:21.972741+00:00","updated_at":"2026-07-04T23:52:21.972741+00:00"}