{"as_of":"2026-08-18T06:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4ffc10df2a44728a9e2e28adfeece549d69ab675ea0a2181363ca4fce08fca51","coverage":[{"denominator":48,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-14T04:54:13.924295Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/1909.02506/citation-record","integrity":"/paper/1909.02506/integrity","json":"/paper/1909.02506/citation-record.json","paper":"/paper/1909.02506"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.509561Z","title":"Schapire","venue":null,"work_id":"644cbc2a-babc-42bf-9c2f-cea4d8c0b61d","year":2017},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.738346Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:fe15e6c1aa9d749699f012c475aa2ae143e8374c19c6ba70133033aed30b73af","observation_id":"2da5c4f8-32bb-4c1c-bdd2-7da15c1aafc3","resolution":{"observed_at":"2026-08-14T04:54:14.513208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.498696Z","title":"Eﬃcient optimal learning for contextual bandits","venue":null,"work_id":"a9894c79-4ab5-4a15-b8c1-5a266623c472","year":2011},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.742927Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:4c37401c82f893a4cde0a32dd8e5138debba665ab4702ef01d647d01c2a14b75","observation_id":"2f559d39-bb65-4437-b035-86153a337c30","resolution":{"observed_at":"2026-08-14T04:54:14.502577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.488333Z","title":"Dynamic programming and optimal control , volume 1","venue":null,"work_id":"25c9ef52-8628-45a7-aa44-43d97816b0ae","year":1995},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.747082Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:77e23375b4bc3814294bc87a77553d0020c220af51f6747fd5e549a5abc5a895","observation_id":"5765909d-2b49-40bf-9eca-ff4ab1211e87","resolution":{"observed_at":"2026-08-14T04:54:14.491944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.477739Z","title":"Approximate Dynamic Programming: Solving the curses of dim ensionality, volume","venue":null,"work_id":"404b5d76-219f-4347-822c-201a42c911a2","year":null},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.751867Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:6ebd9ca2dd049349dfa6ffb267202543f008b4f1b1753ad0809c61cad7241b5c","observation_id":"60f2e34c-8475-4f9c-b361-b942afa71e2e","resolution":{"observed_at":"2026-08-14T04:54:14.481694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.455553Z","title":"Human-level control through deep reinforcement learning","venue":null,"work_id":"5e7f5634-e067-4a91-9126-21a0cf3735e0","year":2015},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.760041Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:a889c87872336354d1ddeda935473cffd021ddd1b6fb3b186dd815c202829f62","observation_id":"9aa0314a-1dfb-46f4-9f31-76c614c0cf63","resolution":{"observed_at":"2026-08-14T04:54:14.459429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.444452Z","title":"Unifying count-based exploration and intrinsic motivation","venue":null,"work_id":"9ed639bf-506c-44e0-a2b5-8a3638afc17b","year":2016},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.764295Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:754481c4bdae3b100fde4e7575e30f0391046d1a904f40f6947352efe79bcc86","observation_id":"a3b5f717-4263-45b2-b6bf-bd38c4a7d0b2","resolution":{"observed_at":"2026-08-14T04:54:14.448483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.433589Z","title":"Mastering the game of go with deep neural networks and tree search","venue":null,"work_id":"7488c811-9571-4127-9bb0-1b8d4c7aa8f0","year":2016},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.768222Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:2df53abbb140052d6e8e38ba37312c1224f225183458f75b9db8df52771148ea","observation_id":"985da6e6-a9ab-442c-ba7e-e40460a9059c","resolution":{"observed_at":"2026-08-14T04:54:14.437345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.422632Z","title":"Ma stering the game of go without human knowledge","venue":null,"work_id":"08d62cba-f23f-4c70-a857-11432b76c86f","year":2017},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.771912Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:8d7871d33e6dd5d9f1aa5ca1542b651f5751af0a313e04eeb7f53c3819002ab8","observation_id":"60df2f6d-336c-4dba-845a-2f26467e27b2","resolution":{"observed_at":"2026-08-14T04:54:14.426430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.411424Z","title":"Deep reinforcement learning for robotic manipulation with asynchronous oﬀ-policy updates","venue":null,"work_id":"da5ed159-454d-4b0f-ac14-c7dd9dddfabb","year":2017},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.775926Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:857b7babf7a79bccf87c2ac208144e268d50a98200f618b25090443ef01abfde","observation_id":"ab1668e3-a944-41da-9d4f-f4008e991b56","resolution":{"observed_at":"2026-08-14T04:54:14.415330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.400192Z","title":"Is Q-learning provably eﬃcient? In Proceedings of Advances in Neural Information Processing S ystems (NeurIPS) , pages 4863–4873,","venue":null,"work_id":"e3e54681-fbae-4c74-be0c-d1a28b21e68f","year":null},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.779697Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:9edc77b52b23a0766e95a28a5290bcb864cb62a6bb7d9ec5b5f82364dc222677","observation_id":"c6d76ad9-7b13-42ce-85ed-7ef61dd5d296","resolution":{"observed_at":"2026-08-14T04:54:14.404304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1901.00210","last_updated":"2019-11-01T20:35:28Z","snapshot_observed_at":"2026-08-14T17:36:09.834261Z","submitted_at":"2019-01-01T21:17:21Z","title":"Tighter Problem-Dependent Regret Bounds in Reinforcement Learning without Domain Knowledge using Value Function Bounds","version":4},"cited_work":{"arxiv_id":"1901.00210","doi":null,"metadata_source":"pith","pith_arxiv_id":"1901.00210","snapshot_observed_at":"2026-08-14T04:54:13.999483Z","title":"Tighter Problem-Dependent Regret Bounds in Reinforcement Learning without Domain Knowledge using Value Function Bounds","venue":"cs.LG","work_id":"9d24c8a0-8562-415a-b027-103df0a9fa52","year":2019},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.784020Z"},"links":{"cited_paper":"/paper/1901.00210","citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:e6914b55d3f31629988fde378744a5d164e855dc8d6bb6447af067dd567a393e","observation_id":"7541a362-e33a-4c63-b5bc-e4617e947fd7","resolution":{"observed_at":"2026-08-14T04:54:14.003851Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.389285Z","title":"Min imax regret bounds for reinforcement learning","venue":null,"work_id":"67a9b100-d9f8-4590-bb1d-90d44b5cc7f7","year":2017},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.788388Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:58373abeb3bcf7bb3fb3b70cc1b51626e54065cb151af1402b4b832044583d2e","observation_id":"c6ff1699-9e4d-4039-a20c-1bfea857f4f2","resolution":{"observed_at":"2026-08-14T04:54:14.393099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.378645Z","title":"Pac model-free reinforcement learning","venue":null,"work_id":"495af090-4c89-4881-8078-83219c86a73e","year":2006},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.792198Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:d012477da6e282053f8a6b85613edc72a04fc011248ebdd10d4d04c472170b58","observation_id":"6457e25c-b60e-4e10-aab5-a021349f3c44","resolution":{"observed_at":"2026-08-14T04:54:14.382243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.367930Z","title":"Speedy q-learning","venue":null,"work_id":"7e50b5f2-8d13-466c-84d6-b6aca1196315","year":2011},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.796245Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:ce167f7ad02dd3f78cc18e2bc980a3c6dd80caf15c6d351627e8e7a503562874","observation_id":"53dad104-a1ff-48bf-b438-5895be4eede5","resolution":{"observed_at":"2026-08-14T04:54:14.371874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.356242Z","title":"Learning rates for q-lear ning","venue":null,"work_id":"411f4d02-bf7c-489d-995a-8cf350f341fc","year":2003},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.800280Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:e8c6ee07385f14469cd6b31625301a334021781d3bcccda7add00244dd98c9e4","observation_id":"3710c4c3-634f-44ba-8bd7-1d9bd8e7c6f3","resolution":{"observed_at":"2026-08-14T04:54:14.360470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.344445Z","title":"Variance re duced value iteration and faster al- gorithms for solving markov decision processes","venue":null,"work_id":"49afcfc9-c307-430d-8daa-03d7e6d8e603","year":null},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.804599Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:e708f2cd9ae01ecb46fa98d91b2cbee71fc71df9bab7ed21bc27b1fee360126f","observation_id":"012aa6c1-db3b-4f91-b3da-ced392c3c3af","resolution":{"observed_at":"2026-08-14T04:54:14.348945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.333701Z","title":"An analysis of bid-price con trols for network revenue manage- ment","venue":null,"work_id":"337b3ebf-10e5-4f66-8cde-9397565b399f","year":1998},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.808675Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:f81a1d7ae93e3ce634819a361e3b9a31f2ff2f96976255df463aeb37cfc734e3","observation_id":"f99b3821-9f17-4986-b7ab-706eb62a1154","resolution":{"observed_at":"2026-08-14T04:54:14.337365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.322643Z","title":"Dynamic bid prices in revenue management","venue":null,"work_id":"146fbab4-384e-4e02-9fb5-68c72f3b119e","year":null},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.812664Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:87aaacf997c38c4b1085d999d8311c8cbfe4d543811a2dcc0437a7e857ee7812","observation_id":"257226e7-a698-4515-b100-85dfb2a514f3","resolution":{"observed_at":"2026-08-14T04:54:14.326612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.311781Z","title":"Asynchronous metho ds for deep reinforcement learning","venue":null,"work_id":"ad024e7e-9302-4132-9edf-a50a00aa0f1c","year":1928},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.816770Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:e5d7687d209bb8b40f3ba01cb106cf4aa897a1e67cbafebbc52397809009949f","observation_id":"b31dc2dc-f577-4717-b60e-64d439a354da","resolution":{"observed_at":"2026-08-14T04:54:14.315679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.300872Z","title":"The malmo platform for arti- ﬁcial intelligence experimentation","venue":null,"work_id":"a337d694-e566-48d1-be1e-8d1181a765b8","year":2016},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.820740Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:bb865e62a2d414e8b0f4cb927302e297f5fdb112071dcfcc976ba9c99bde2094","observation_id":"4a48be75-ec48-4686-883c-ce9ec514f066","resolution":{"observed_at":"2026-08-14T04:54:14.304884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.289517Z","title":"Gambling in a rigged casino: The adversarial multi-armed bandit problem","venue":null,"work_id":"d6c0b610-b3a6-46c8-b3bd-841066c8dc50","year":1995},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.825045Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:81e22d2acf43d1dd0191bf88232525408108bf9e9c5d1142c60d0e1c514d29bb","observation_id":"a09365b0-df07-499c-9f80-b48ac7c68536","resolution":{"observed_at":"2026-08-14T04:54:14.293282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.278508Z","title":"Learning from delayed rewards","venue":null,"work_id":"f1d46ac8-e308-4809-8d0b-b24f5528cf8b","year":1989},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.829012Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:efab07eb9d98f6d6b31b95f5d9539383526f79beb7a549f3c8367e52aeaf6d05","observation_id":"0aa6861f-a9e4-4934-9c9d-2c2a5a5c2caa","resolution":{"observed_at":"2026-08-14T04:54:14.282303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.267149Z","title":"Q-learning","venue":null,"work_id":"d1fadcfc-ca51-47aa-a9bf-a98fefb4dbcd","year":1992},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.832613Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:e09e503e54e48be0df597d0b1e1562da6468e8c426b5e2ea5f798d6bc37bc8a7","observation_id":"a9b4beea-ee82-4b7a-9ff5-064f5027c09f","resolution":{"observed_at":"2026-08-14T04:54:14.271279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.254986Z","title":"Asynchronous stochastic approximation and q -learning","venue":null,"work_id":"50fabe59-e59f-40c3-83e5-89aa2bc18ed5","year":1994},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.836099Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:57c564c6a39fbef6f1090307eb6f713a61551c76fae4f4be5136169a097a777e","observation_id":"543ac82f-4627-4459-b740-67a81400fc24","resolution":{"observed_at":"2026-08-14T04:54:14.259376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.242201Z","title":"Regu- larized policy iteration with nonparametric function spaces","venue":null,"work_id":"ccd688dd-2c23-458f-8286-a05315b29ccf","year":2016},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.839619Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:9083db0d5137376bc64bfac440303359255827af375fd0b978e6e29f3874dfc3","observation_id":"fa802055-7200-4088-b23e-ee942157aadc","resolution":{"observed_at":"2026-08-14T04:54:14.247160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.230745Z","title":"Finite-sample analysis of least- squares policy iteration","venue":null,"work_id":"8ef05b2a-3ed3-40f4-9b5f-8b0ce4c1f331","year":2012},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.843708Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:84a065e19de6a2497a17bfe8c1bb1e98e79f98ef53638a364230f5728e92c7b0","observation_id":"be1358dd-9d80-4b2a-aa15-aa4a969a9253","resolution":{"observed_at":"2026-08-14T04:54:14.234671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.219050Z","title":"Learning near-optimal policies with bellman-residual minimization based ﬁtted policy iteration and a single sample path","venue":null,"work_id":"7745b9e8-9b71-4372-8306-3dc51b731694","year":null},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.847618Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:a52ef23fa55aed6c7b6e57d5c094ebdd4c1a5b4a78a7e3f3efd93e394c163059","observation_id":"1bdd4641-48d3-470b-b6f8-d777a05c04ac","resolution":{"observed_at":"2026-08-14T04:54:14.223709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.207940Z","title":"Finite-time bounds for ﬁt ted value iteration","venue":null,"work_id":"f0f62ff7-cc67-44a8-b721-d095b0116e33","year":2008},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.851775Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:c3835c4d862dbbcfdd4fd445791b0d90dc22c18f58dfd71a0fb5f6a7e1847ee5","observation_id":"3ada3838-90ba-47e3-a4c1-a75e3b3722d7","resolution":{"observed_at":"2026-08-14T04:54:14.212025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.196294Z","title":"Information-theoretic consideratio ns in batch reinforcement learning","venue":null,"work_id":"ec8ace66-0369-4557-a4e3-98400ff28988","year":2019},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.855880Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:05e6cc55e11a5bf4e789a341c87c8c42f31a259f61b4238898b5209b64269b2c","observation_id":"056c1862-9c91-4e10-a7de-62d4e0d4e10f","resolution":{"observed_at":"2026-08-14T04:54:14.200101Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.183852Z","title":"Stable function approximation in dynamic pro gramming","venue":null,"work_id":"2faf94f9-d205-4c66-9fd5-161a7189c09a","year":1995},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.859818Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:1f86a3629b81bbe596ea6e271bc75dbade682e0789c01299fc6411741fc68b2b","observation_id":"0002a454-f72e-466d-ac60-e94f78eff01d","resolution":{"observed_at":"2026-08-14T04:54:14.188519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.05388","last_updated":"2019-08-08T07:23:12Z","snapshot_observed_at":"2026-08-17T23:29:59.724188Z","submitted_at":"2019-07-11T17:06:11Z","title":"Provably Efficient Reinforcement Learning with Linear Function Approximation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.05388","snapshot_observed_at":"2026-08-14T04:54:13.863800Z","title":"P rovably eﬃcient reinforcement learning with linear function approximation","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.863800Z"},"links":{"cited_paper":"/paper/1907.05388","citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:3691128e2546bba0779dfbbd0ff7598e33c7ffb6c277b0296ea44abe1749ac12","observation_id":"70db911d-5bc8-49d7-9b49-3b148c7fdae3","resolution":{"observed_at":"2026-08-14T04:54:13.863800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.171181Z","title":"On oracle-eﬃcient pac rl with rich observations","venue":null,"work_id":"6a48534a-005a-4e0d-8f39-3e1f78aac007","year":2018},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.868211Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:38a7487c9f0133742c9c13c484537cf469d02b5f9bfe1ff85e81a8c8a2f484f4","observation_id":"6d6e05e9-b7c9-4ff2-abcb-f222d4938260","resolution":{"observed_at":"2026-08-14T04:54:14.176506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.159524Z","title":"Du, Akshay Krishnamurthy, Nan Jiang, Alekh Agarwal, M iroslav Dudik, and John Langford","venue":null,"work_id":"3c97f8f5-fd65-451d-8c92-c65f13e918b6","year":2019},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.871945Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:8dc321105e11891e5140c21301dc26098a564745858e16ae01c6a8dfdbc6d080","observation_id":"c6df8ebc-569a-4987-95c3-27c4c4c98a16","resolution":{"observed_at":"2026-08-14T04:54:14.163682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.147654Z","title":"Eﬃcient reinforcement learnin g in deterministic systems with value function generalization","venue":null,"work_id":"538fd6a0-192a-45c6-9151-0751c7b303b8","year":2017},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.875576Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:0c87eae22dda9bd23c7ac2cbab7b72c22892ed163f6cddb642107e58660679ed","observation_id":"96047cdd-2dd3-4e8d-b827-f1a3f13fc13c","resolution":{"observed_at":"2026-08-14T04:54:14.151711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.133763Z","title":"N ear-optimal time and sample complexities for solving discounted markov decision process with a ge nerative model","venue":null,"work_id":"adbb0fbc-601a-4eb5-b296-bbec3cf8fa07","year":2018},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.879925Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:428f11afb50b14704ed24315f40be964cbe82cc35037098bb0466ff6d207fe35","observation_id":"4ea3b8f1-d38c-4c25-8cb4-a165513d1543","resolution":{"observed_at":"2026-08-14T04:54:14.138620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.120262Z","title":"Glo bal convergence of policy gradient methods for linearized control problems","venue":null,"work_id":"f9734b4f-58ed-4998-8818-013fb713d275","year":2018},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.883514Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:2d410564be2365541b7b3c89fb92c7a7835024941562105b2247d4aba8056ef9","observation_id":"7efce2da-627c-4d07-a012-eb34ddf29243","resolution":{"observed_at":"2026-08-14T04:54:14.125486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.06265","last_updated":"2019-06-24T17:09:55Z","snapshot_observed_at":"2026-08-16T08:39:16.101139Z","submitted_at":"2019-05-15T16:06:30Z","title":"Stochastic approximation with cone-contractive operators: Sharp $\\ell_\\infty$-bounds for $Q$-learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.06265","snapshot_observed_at":"2026-08-14T04:54:13.886912Z","title":"Stochastic approximation with cone-contr active operators: Sharp ℓ∞-bounds for q-learning","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.886912Z"},"links":{"cited_paper":"/paper/1905.06265","citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:20d6f22862712cac460ea6a72ed7ee3520a0ea89c6a6e4b065ab845e289c73ca","observation_id":"48344bfe-79b9-435f-a0a5-561279b0772f","resolution":{"observed_at":"2026-08-14T04:54:13.886912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.108284Z","title":"Taming the monster: A fast and simple algorithm for contextual bandits","venue":null,"work_id":"3761b13e-70bb-4bdb-934f-0b026068490a","year":2014},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.890641Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:2ac02aef7116c1091493b83e4f1f9b194869ad5b348fe19fca0293fb6bfd6e68","observation_id":"b7573e80-b58a-4011-874b-6b5b620b634d","resolution":{"observed_at":"2026-08-14T04:54:14.112407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.10389","last_updated":"2019-06-13T07:28:19Z","snapshot_observed_at":"2026-08-14T16:26:57.167418Z","submitted_at":"2019-05-24T18:02:39Z","title":"Reinforcement Learning in Feature Space: Matrix Bandit, Kernels, and Regret Bound","version":2},"cited_work":{"arxiv_id":"1905.10389","doi":null,"metadata_source":"pith","pith_arxiv_id":"1905.10389","snapshot_observed_at":"2026-08-14T04:54:13.957214Z","title":"Reinforcement Learning in Feature Space: Matrix Bandit, Kernels, and Regret Bound","venue":"cs.LG","work_id":"3703f63b-ac58-4045-8d01-ad0af9e31e65","year":2019},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.894260Z"},"links":{"cited_paper":"/paper/1905.10389","citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:19bf7ec35f1949b6a1355a20c16c5fe701685009297f09a3473defdffef93606","observation_id":"d86242a7-485e-42aa-905c-954229a22d59","resolution":{"observed_at":"2026-08-14T04:54:13.963706Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.097919Z","title":"On learning sets and functions","venue":null,"work_id":"1cfc63d6-32d1-41b3-b2b3-4c7f5d2c87ba","year":1989},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.897850Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:18a9035864d9f75fda4d9b78131f4f8b5b783ab1593d095d2e6ad85e206f97ff","observation_id":"51440e06-2d73-446b-9d59-e3b16f2cbdea","resolution":{"observed_at":"2026-08-14T04:54:14.101367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.087002Z","title":"Decision theoretic generalizations of the pac mo del for neural net and other learning applications","venue":null,"work_id":"7c8ded25-63c3-4b4c-a5e4-5b553de83678","year":1992},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.901592Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:a4a3a0a9721400bd2e098e42b51b9827a98883a90da15a9b9c4bc21662987651","observation_id":"d12b1613-a872-4dc4-b587-ae450c1e7b5c","resolution":{"observed_at":"2026-08-14T04:54:14.090701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.075414Z","title":"Rates of convergence in the central limit the orem for empirical processes","venue":null,"work_id":"23ebf5d6-4efa-4253-8c44-ca0d11ec9925","year":1986},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.905224Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:ecb6d8f5a1822a9c2e651e3fd4405b7d16565a4df5a824b62b319dd543f9023d","observation_id":"05a60069-8a0e-4796-9b03-781f3fdfc24a","resolution":{"observed_at":"2026-08-14T04:54:14.080068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.063585Z","title":"On general minimax theorems","venue":null,"work_id":"4b3fdb91-4d9b-425e-b28a-c0a0f1947797","year":1958},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.909420Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:3b73c0a5d464003a435cf9e15614fc5ecd94c3886ad677fd84682cc08352b389","observation_id":"c2f41ed6-565d-470f-a3d2-221fc359dfa0","resolution":{"observed_at":"2026-08-14T04:54:14.067730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.049438Z","title":"On minimum volume ellipsoids containing part of a give n ellipsoid","venue":null,"work_id":"5bbf9635-98d4-455d-b901-fd4002a40834","year":1982},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.913339Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:a8b13419612e05a00a8343c1c138993c5736ddd16b272daea85eb5a16236305a","observation_id":"4eb87988-5959-4ab2-b515-148935a61a47","resolution":{"observed_at":"2026-08-14T04:54:14.054233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.036294Z","title":"Sphere packing numbers for subsets of the bo olean n-cube with bounded vapnik- chervonenkis dimension","venue":null,"work_id":"9396a92c-9f5d-4e11-83f1-bf61ff75bdc5","year":1995},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.916872Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:009b493f3e65de96a8a79425e60bf09ff1a9f454b11be57089497cd965bcaa08","observation_id":"c7aa974c-3be4-41e0-9f41-d2f8ae213fef","resolution":{"observed_at":"2026-08-14T04:54:14.040507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.023828Z","title":"Springer Science & Business Media, 2013","venue":null,"work_id":"f4bd1213-07a6-46a8-8e91-c4e770952798","year":2013},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.920614Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:a2aed177d684e39da1961b8ba72e1f2909add952210f077d13327c7bb4f8aa85","observation_id":"2c84bc67-97d4-4741-84f9-54cdb046dbd1","resolution":{"observed_at":"2026-08-14T04:54:14.027802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.012192Z","title":"Convergence of stochastic processes","venue":null,"work_id":"12ae7845-5b8b-497c-95d7-7787ba516b42","year":2012},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.924295Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:88ab652ccfa63846f6fc5344a7074db3d5f6654b78b959235141c2b3306b2a44","observation_id":"aced5f6a-38ae-4bb1-adfe-07eddab1dd2a","resolution":{"observed_at":"2026-08-14T04:54:14.015789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:54:14.466753Z","title":null,"venue":null,"work_id":"5d57b769-8997-4ea8-be2d-b04d0f4c0a78","year":2007},"citing_paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank","version":3},"reference_index":703,"source":"pdf_text","source_observed_at":"2026-08-14T04:54:13.755928Z"},"links":{"citing_paper":"/paper/1909.02506"},"observation_digest":"sha256:0bb56609d8ec0a9587a40e5006256a47f753c50ebac1270391ac0d0c890424da","observation_id":"a1d8be10-b6d3-4b4e-9381-4490a608b6b0","resolution":{"observed_at":"2026-08-14T04:54:14.470456Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"1909.02506","last_updated":"2020-06-20T17:17:19Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-16T03:49:01.484422Z","submitted_at":"2019-09-05T16:20:41Z","title":"$\\sqrt{n}$-Regret for Learning in Markov Decision Processes with Function Approximation and Low Bellman Rank"},"reference_resolution":{"displayed":48,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":3,"verified_exact":1,"verified_fuzzy":43},"total_outbound_references":48},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 48 of 48 outbound references and 0 inbound Pith citation observations for arXiv:1909.02506."}