{"as_of":"2026-08-11T08:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:86a69d4ae02ed88fe69e6321771ae8d93f71ae5be1bb972615d483209941c60d","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T16:18:17.535895Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.13394/citation-record","integrity":"/paper/2501.13394/integrity","json":"/paper/2501.13394/citation-record.json","paper":"/paper/2501.13394"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.332912Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.332912Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:1cae4c37d9596f8b782b650d6e77033db6fffc20eeaf18420fa7b01fcf2b7596","observation_id":"02eb01fd-2b88-409d-b363-75ffe4affd28","resolution":{"observed_at":"2026-08-10T16:18:17.332912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.195964Z","title":"H., Chalup, S","venue":null,"work_id":"ebb96672-ade8-4a8d-9465-8d6fbafc9672","year":2015},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.337507Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:246421887d340f18824425443c907bba7194842edebd0a1245a8b28f829ae78b","observation_id":"0099de47-d04d-4054-af78-4e0ff8b8a05e","resolution":{"observed_at":"2026-08-10T16:18:18.199357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.03966","last_updated":"2017-05-09T14:41:15Z","snapshot_observed_at":"2026-07-06T04:59:45.260451Z","submitted_at":"2016-06-13T14:17:00Z","title":"Making Contextual Decisions with Low Technical Debt","version":2},"cited_work":{"arxiv_id":"1606.03966","doi":null,"metadata_source":"pith","pith_arxiv_id":"1606.03966","snapshot_observed_at":"2026-08-10T16:18:17.759211Z","title":"Making Contextual Decisions with Low Technical Debt","venue":"cs.LG","work_id":"516741bc-dbfd-4b5a-baac-19aa6122bd1d","year":2016},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.341200Z"},"links":{"cited_paper":"/paper/1606.03966","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:ab500d2cbd4e43fda358089fd793c0d2e8f8cd8607b8e27b59019e73c0ed8ce4","observation_id":"bc7267e0-51fc-4c9c-8901-70f6b521745d","resolution":{"observed_at":"2026-08-10T16:18:17.762973Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.186673Z","title":"M., Lee, J","venue":null,"work_id":"a18ef749-b4c5-4291-8094-b172a3908490","year":2020},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.345338Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:d88c618ec9d033427da1444cbfe9fd0d429bb3ffd17359cb765dfb8a3d550eb9","observation_id":"dcc3fb88-bbae-485f-8aa7-92464c32d0e5","resolution":{"observed_at":"2026-08-10T16:18:18.189991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.177529Z","title":"Improved worst-case regret bounds for randomized least-squares value iteration","venue":null,"work_id":"1503aadb-bb84-44df-931b-86a2a14319a0","year":2021},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.348745Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:80c033b9a2e054a8aa2a588c1dcaff64770f48e152ea2d0b4c4c2d185209db1f","observation_id":"5782fdc6-b790-4eef-8a3d-6007ed2115e4","resolution":{"observed_at":"2026-08-10T16:18:18.180779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.352216Z","title":"Near-optimal regret bounds for reinforcement learning","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.352216Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:19d9f3aa11bb85a09f2b26816b77246d9bbf97bc04de0e8b7d58b381fd065e48","observation_id":"4f61e943-ee00-489e-9aaa-7b68e1a8f149","resolution":{"observed_at":"2026-08-10T16:18:17.352216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.162636Z","title":"G., Munos, R., Ghavamzadeh, M., and Kappen, H","venue":null,"work_id":"1e7c2787-92c1-4853-81fd-214340a4d43f","year":2011},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.355443Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:79e67da9e8e93af182d3becbca31a3f4f9c4ec3a2b90169510cbb70eeb1f207e","observation_id":"68359b24-00d1-4ac5-92fc-e636ee94386f","resolution":{"observed_at":"2026-08-10T16:18:18.165845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1710.03748","last_updated":"2018-03-14T21:09:49Z","snapshot_observed_at":"2026-07-06T06:03:37.247241Z","submitted_at":"2017-10-10T17:59:41Z","title":"Emergent Complexity via Multi-Agent Competition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1710.03748","snapshot_observed_at":"2026-08-10T16:18:17.358850Z","title":"Emergent complexity via multi-agent competition","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.358850Z"},"links":{"cited_paper":"/paper/1710.03748","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:d8dd1d6440345c502e81df9fdddd30fea096dbe26e894cd7b9b6a1242230c8f5","observation_id":"d29dac7b-09fe-40c3-ad64-6851b57972ed","resolution":{"observed_at":"2026-08-10T16:18:17.358850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1205.2661","last_updated":"2012-05-09T14:47:06Z","snapshot_observed_at":"2026-07-06T02:47:58.266745Z","submitted_at":"2012-05-09T14:47:06Z","title":"REGAL: A Regularization based Algorithm for Reinforcement Learning in Weakly Communicating MDPs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1205.2661","snapshot_observed_at":"2026-08-10T16:18:17.362465Z","title":null,"venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.362465Z"},"links":{"cited_paper":"/paper/1205.2661","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:879d32dc19aeef0d9efc8ca2684b16183ad84102993a9375d636a824cb7c4790","observation_id":"c9c514a6-8f8d-4fec-a650-1e9eea0d3f66","resolution":{"observed_at":"2026-08-10T16:18:17.362465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.153679Z","title":"Multi-agent reinforcement learning: An overview","venue":null,"work_id":"bdf25b07-7034-413d-aec6-4b9ee63cf049","year":2010},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.366014Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:93ef2b712e8f653380151ef248bf201d9d84d067a47b0f34610a8c09d9fc53dd","observation_id":"823456aa-af2e-4d52-804f-95c46703b470","resolution":{"observed_at":"2026-08-10T16:18:18.156934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.144453Z","title":"Society of agents: Regret bounds of concurrent thompson sampling","venue":null,"work_id":"bd2d0995-1677-47a4-9a57-b535e67609dd","year":2022},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.369418Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:c1af2b72b23b8534c01da882e9dbe04074f1bd5c7418164d6c83f328f7adbf8a","observation_id":"38817715-be49-4ea1-b7c0-cc3ec889da31","resolution":{"observed_at":"2026-08-10T16:18:18.147754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.135238Z","title":"A provably efficient model-free posterior sampling method for episodic reinforcement learning","venue":null,"work_id":"8d5fd285-8226-4719-9c5e-723c0ba2e8b4","year":2021},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.372932Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:f27a927e8569d8d33b6457d16c2c41189ce1501a54c213dd7f2c9a1e6ce36416","observation_id":"452dc6ca-3001-4a22-bcd9-9255736d7a60","resolution":{"observed_at":"2026-08-10T16:18:18.138606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.125984Z","title":null,"venue":null,"work_id":"6e4bb03c-03ae-40c8-b70f-e485a85a0862","year":2018},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.376237Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:4d1f8cdce6f173669094e9e691d51d6228b2c1854bb0818eb0207a6f1d5843aa","observation_id":"9d9ec6c9-c161-46b4-96ff-5db588d2ef22","resolution":{"observed_at":"2026-08-10T16:18:18.129176Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.116418Z","title":"and Van Roy, B","venue":null,"work_id":"59844479-d3fb-405d-afb0-37912da507f7","year":2018},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.379428Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:a83b7b09d3dc495af0a574b121120cbcb70d3ab50b3bed86b46e01dc7edd1a4c","observation_id":"78d5a7a3-a4e5-434a-974a-07e4a671e27c","resolution":{"observed_at":"2026-08-10T16:18:18.119587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.106944Z","title":"Scalable coordinated exploration in concurrent reinforcement learning","venue":null,"work_id":"fbbb33f5-dbb4-4e89-9779-cd6b3805232c","year":2018},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.382882Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:3bfa70a6b34b6046a9de74a8c0f0f34ec70e92ab141a4c9f8221fd8a3956ec85","observation_id":"dbb5718e-7d74-47db-bd6d-42b098e06ad7","resolution":{"observed_at":"2026-08-10T16:18:18.110347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1901.09311","last_updated":"2019-09-27T02:09:54Z","snapshot_observed_at":"2026-07-06T07:29:14.938525Z","submitted_at":"2019-01-27T03:44:42Z","title":"Q-learning with UCB Exploration is Sample Efficient for Infinite-Horizon MDP","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1901.09311","snapshot_observed_at":"2026-08-10T16:18:17.386389Z","title":"Q -learning with UCB exploration is sample efficient for infinite-horizon MDP","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.386389Z"},"links":{"cited_paper":"/paper/1901.09311","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:41ff4bfc41396697953c494f8bdcb6ac9be452cf8000903f42b3c322f58ae863","observation_id":"e73cbf13-dca9-4c8c-b6d9-743b092cd34e","resolution":{"observed_at":"2026-08-10T16:18:17.386389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.06366","last_updated":"2020-02-19T06:05:39Z","snapshot_observed_at":"2026-08-07T06:08:50.758806Z","submitted_at":"2019-12-13T09:10:18Z","title":"Provably Efficient Reinforcement Learning with Aggregated States","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.06366","snapshot_observed_at":"2026-08-10T16:18:17.390316Z","title":"Provably efficient reinforcement learning with aggregated states","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.390316Z"},"links":{"cited_paper":"/paper/1912.06366","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:ada55d1c197051bffb30b3fe840af8fc00b4e92b50200290964ba8bc8ad3a459","observation_id":"6b58b1e7-c05e-43d9-bd33-fbb2f0f2c266","resolution":{"observed_at":"2026-08-10T16:18:17.390316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.095997Z","title":"Simple agent, complex environment: Efficient reinforcement learning with agent states","venue":null,"work_id":"af0d6ad9-469a-4a57-b842-2c33424b303e","year":2022},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.394064Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:52f989f40b8a880b81b81760514ae6ab7e9a905a2c7bbb01c2be68a7347da961","observation_id":"51f45872-1e53-4662-9974-3a31d97f8e04","resolution":{"observed_at":"2026-08-10T16:18:18.100341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.04972","last_updated":"2021-03-08T18:51:00Z","snapshot_observed_at":"2026-08-08T15:11:36.603603Z","submitted_at":"2021-03-08T18:51:00Z","title":"Provably Efficient Cooperative Multi-Agent Reinforcement Learning with Function Approximation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.04972","snapshot_observed_at":"2026-08-10T16:18:17.397466Z","title":"and Pentland, A","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.397466Z"},"links":{"cited_paper":"/paper/2103.04972","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:307f1d02ee64cb8383d2d74e35bb282dae9ecf19ab8622f252dd299ae68797c8","observation_id":"8d3f16d0-dfa2-4eeb-9ab3-ca1f455cff8c","resolution":{"observed_at":"2026-08-10T16:18:17.397466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.07464","last_updated":"2020-06-12T20:59:21Z","snapshot_observed_at":"2026-08-05T02:44:03.694223Z","submitted_at":"2020-06-12T20:59:21Z","title":"Hypermodels for Exploration","version":1},"cited_work":{"arxiv_id":"2006.07464","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.07464","snapshot_observed_at":"2026-08-10T16:18:17.698120Z","title":"Hypermodels for Exploration","venue":"cs.LG","work_id":"c5daa247-5aa0-4a07-b0a0-c778ff8d6745","year":2020},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.401271Z"},"links":{"cited_paper":"/paper/2006.07464","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:4447edcb2310820f4f9ca106f0a92885f5dbdff267f5db85a09024aab9ee4778","observation_id":"b4f6bb5b-e54b-46a1-b1fd-8f708fa31cae","resolution":{"observed_at":"2026-08-10T16:18:17.702689Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.086371Z","title":"Bayesian bellman operators","venue":null,"work_id":"5946cf5b-d23b-4359-a9b3-4078843d9528","year":2021},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.404910Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:1ab7129884bd049ee29c65d55a6b995cfa8e18013d8de0b139ceee943e048fd2","observation_id":"8cada472-bec8-42f6-809f-683292db9677","resolution":{"observed_at":"2026-08-10T16:18:18.089595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.076877Z","title":"Deep reinforcement learning for robotic manipulation with asynchronous off-policy updates","venue":null,"work_id":"89458132-4070-4194-af70-551eb24eb469","year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.408275Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:4a1e46d860db2f52c173f3434464403b01ffe7919a6c9215e21e129419a08fdf","observation_id":"3963d48d-2ad9-4107-bcab-c9eb78fb4c3a","resolution":{"observed_at":"2026-08-10T16:18:18.080301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.067229Z","title":"and Brunskill, E","venue":null,"work_id":"c4914aba-38b1-4226-8106-29636ac593b4","year":2015},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.411736Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:2a7b430b68720c31c9c78f9a958016f2bcb8f2bbe4e6a24ef09b515bb3cd5189","observation_id":"31e57f91-8593-4f52-87bf-0f91b49035b7","resolution":{"observed_at":"2026-08-10T16:18:18.070456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.057541Z","title":"Randomized exploration in reinforcement learning with general value function approximation","venue":null,"work_id":"7ade9224-d9d7-4fff-ae8d-9fe6a80b5279","year":2021},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.415159Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:c496d4af1d462b5dba902829f0931b544cd02526f236638dc3f8bf2dc4144a8e","observation_id":"06d4e5e6-a1ae-41c1-86e2-3f48e1a06125","resolution":{"observed_at":"2026-08-10T16:18:18.060969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18246","last_updated":"2024-03-18T00:37:12Z","snapshot_observed_at":"2026-08-10T23:54:04.308354Z","submitted_at":"2023-05-29T17:11:28Z","title":"Provable and Practical: Efficient Exploration in Reinforcement Learning via Langevin Monte Carlo","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18246","snapshot_observed_at":"2026-08-10T16:18:17.418546Z","title":"R., Precup, D., Anandkumar, A., and Azizzadenesheli, K","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.418546Z"},"links":{"cited_paper":"/paper/2305.18246","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:4e5814c8776c16871be11d73e34f20d58ef64146dd22caeb0027b41ad47798ac","observation_id":"8d903ea8-6c0a-4264-a522-401f8c096966","resolution":{"observed_at":"2026-08-10T16:18:17.418546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.047383Z","title":"M., and Tschiatschek, S","venue":null,"work_id":"952792e1-391f-4ef8-86cf-af3f5400b774","year":2019},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.421996Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:1acb1832ce721a4d4f9a031ccf46ed5fb6cd4f9fbd12fb005e091cbb54aeca5a","observation_id":"b87360a2-93b8-400e-98b9-56edf27c755d","resolution":{"observed_at":"2026-08-10T16:18:18.051084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.425392Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.425392Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:f9c1841adf281f1e8565616e890be84fe9c2b13f3b33d07694edddb770ac6926","observation_id":"de87b722-3feb-448a-9597-7df7c424a35f","resolution":{"observed_at":"2026-08-10T16:18:17.425392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.429589Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.429589Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:4d1e370112df49b9c551f31965620924575466107f659f5215ca7ea5351966a7","observation_id":"19ff4a63-3073-49a5-aecd-d1b9d64f2996","resolution":{"observed_at":"2026-08-10T16:18:17.429589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.025800Z","title":null,"venue":null,"work_id":"8f68920e-51c0-4dd6-bde4-960adcfe1c23","year":2020},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.433136Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:7d03f6eff3f8e988ebef6ec765a99191874c22800d6daa68cdb585ecb498b52f","observation_id":"74563cfa-979c-4edb-91bc-ebf2227c29bc","resolution":{"observed_at":"2026-08-10T16:18:18.029056Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.015192Z","title":null,"venue":null,"work_id":"c88df9d1-7310-48ee-9ecb-11741bfc71cf","year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.436300Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:aac9a61edc5edc9e4cdc8d864b62f41c77a015e1696756c9073a8b38c9e79c39","observation_id":"13d21b87-8b4d-4b15-af11-bac77edc4924","resolution":{"observed_at":"2026-08-10T16:18:18.018562Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:18.004937Z","title":null,"venue":null,"work_id":"40a3fabe-93d7-4042-af80-89ce63e2a6bf","year":2010},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.439544Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:954eb7891a6666f2ca0a0b6e8f16ce11e9574097610475e4a791693ac6362363","observation_id":"2e967ad5-0af7-4832-ba2a-cfcf81017074","resolution":{"observed_at":"2026-08-10T16:18:18.008512Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.994312Z","title":null,"venue":null,"work_id":"946fd685-8f38-48ae-8b0e-f3dce7fc76f4","year":2001},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.442674Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:738f70218e38faa049befc27bbc0abc39a2e35dc1a142fdffca50f97ab56b33f","observation_id":"a3288c5d-451a-4f9c-95da-3a726153e7ce","resolution":{"observed_at":"2026-08-10T16:18:17.997834Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.445834Z","title":"I., Tamar, A., Harb, J., Pieter Abbeel, O., and Mordatch, I","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.445834Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:a0bcd4bc9528dbd65b6acfd86164f9e3eecea6949320460163374572b7722ce4","observation_id":"1db6888e-1b75-44a5-8019-4fe0a6ac4515","resolution":{"observed_at":"2026-08-10T16:18:17.445834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.978110Z","title":"Cooperative multi-agent reinforcement learning: Asynchronous communication and linear function approximation","venue":null,"work_id":"fe47a793-f31f-45f0-af71-0fe555d2732a","year":2023},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.448976Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:db6bcbf4f4c12f761da95ec0c5d48140f24c9cbd438cf02cad174b63fecb6348","observation_id":"018d260c-4c7e-4756-b50c-4a00861a092d","resolution":{"observed_at":"2026-08-10T16:18:17.981797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.968043Z","title":"and Van Roy, B","venue":null,"work_id":"5506ceea-a0ba-42ea-9111-27eedb51509f","year":2014},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.452154Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:13802e8861e6053c27ba03af2b3a10ab843d31451aa6bbec4b38839dbfc96405","observation_id":"d56e2646-40ab-45f0-a543-b97324af78de","resolution":{"observed_at":"2026-08-10T16:18:17.971547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1706.04241","last_updated":"2017-06-13T20:22:54Z","snapshot_observed_at":"2026-07-06T05:46:43.820045Z","submitted_at":"2017-06-13T20:22:54Z","title":"On Optimistic versus Randomized Exploration in Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"1706.04241","doi":null,"metadata_source":"pith","pith_arxiv_id":"1706.04241","snapshot_observed_at":"2026-08-10T16:18:17.670948Z","title":"On Optimistic versus Randomized Exploration in Reinforcement Learning","venue":"stat.ML","work_id":"bcfebce1-5c7c-4477-af97-fc85436ab91b","year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.455424Z"},"links":{"cited_paper":"/paper/1706.04241","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:3561b7730eff2a7a94fb651cba283b33a34542041476bcf5dc28c1905bc88c0f","observation_id":"7af09584-591b-402a-9623-62dea23801c6","resolution":{"observed_at":"2026-08-10T16:18:17.675088Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.957841Z","title":"( M ore) efficient reinforcement learning via posterior sampling","venue":null,"work_id":"101e6d15-9479-4965-bff9-848f70f7a614","year":2013},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.458802Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:a1cf0c21b6db14a35b8d1a2fe3037ce6ae89f6ef774bd69cea93e8b26837c136","observation_id":"d8e80065-e474-4124-8c97-84c31898d9e2","resolution":{"observed_at":"2026-08-10T16:18:17.961243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.947462Z","title":"Deep exploration via bootstrapped DQN","venue":null,"work_id":"2e087bae-8212-4408-a56e-0aa02a52f9ed","year":2016},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.462000Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:5c6833f6ba85248a590289b75ebb32fe3a4a572e0a665116bd5f6cc3b43634ab","observation_id":"edb6352d-6f2d-4a3e-beb1-dbcfbaeb60fa","resolution":{"observed_at":"2026-08-10T16:18:17.951060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.937520Z","title":"J., Wen, Z., et al","venue":null,"work_id":"05270591-76d0-42ef-a295-babad4d7c29b","year":2019},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.465163Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:18932732e33409b9a533c4263e3d9a917ec5600704b45e69243cdc1030ad9cab","observation_id":"11459e4e-5355-41db-9719-3281d0c2d48b","resolution":{"observed_at":"2026-08-10T16:18:17.940926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.927501Z","title":"and Parr, R","venue":null,"work_id":"b8ee63cb-7f46-47db-b934-6ecc49349788","year":2016},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.468243Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:8ce883cd8dc851cce175608646a380b4555ba919fda767b448a5d5cb5fe62f70","observation_id":"17e9035b-6bf4-4019-bcad-613d8f64e534","resolution":{"observed_at":"2026-08-10T16:18:17.931009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.917288Z","title":"Worst-case regret bounds for exploration via randomized value functions","venue":null,"work_id":"8886b7b6-8642-426c-b99c-1da6fe3cb93d","year":2019},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.471356Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:63bbf416b98e86ec41235716867a987fba2ed37b4038b87f16ea28e918b30b09","observation_id":"354e619f-3119-46f3-b542-e467fde92d3b","resolution":{"observed_at":"2026-08-10T16:18:17.920774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.474577Z","title":"and Van Roy, B","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.474577Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:b9c7f71d32a8976565e2331bde9c572e64d12f7ad1c7337e9c084585c6516350","observation_id":"5ac2be83-181e-4420-8550-a36cf35dd0f5","resolution":{"observed_at":"2026-08-10T16:18:17.474577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.06890","last_updated":"2021-01-18T05:52:22Z","snapshot_observed_at":"2026-08-08T12:15:59.639463Z","submitted_at":"2021-01-18T05:52:22Z","title":"Cooperative and Competitive Biases for Multi-Agent Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2101.06890","doi":null,"metadata_source":"pith","pith_arxiv_id":"2101.06890","snapshot_observed_at":"2026-08-10T16:18:17.656985Z","title":"Cooperative and Competitive Biases for Multi-Agent Reinforcement Learning","venue":"cs.LG","work_id":"3eb3449c-0b1b-42f3-bf29-6f5f6de66494","year":2021},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.477793Z"},"links":{"cited_paper":"/paper/2101.06890","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:1048ee06b2b8bbfe4348dc11e1ab66abe28db5aa5f0d1aa348d3e5a6e7c6c162","observation_id":"221eb884-702a-4c82-895b-e8399fb69849","resolution":{"observed_at":"2026-08-10T16:18:17.660714Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.901004Z","title":"and Leyton-Brown, K","venue":null,"work_id":"289dc581-1003-4c36-b1a7-03bef74b0677","year":2008},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.481692Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:2896167400245bdd0a3664de9911b9ab11f76a2fcc48fba3f8090843e8d08eeb","observation_id":"156f37e6-1807-42d9-a702-88bb985e3d3c","resolution":{"observed_at":"2026-08-10T16:18:17.904574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.891005Z","title":"Multi-agent reinforcement learning: a critical survey","venue":null,"work_id":"3141330e-d6aa-4996-b61f-68155129be8e","year":2003},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.484804Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:3c57c8294a51d7524c508e17bb0cc4ce0c6535726a21dfb6089041085724c5f4","observation_id":"efffe54a-82a3-44c7-9f90-6ada333971f2","resolution":{"observed_at":"2026-08-10T16:18:17.894475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.881123Z","title":"Concurrent reinforcement learning from customer interactions","venue":null,"work_id":"c0a9d946-5e83-42f8-8ee9-76d7d97d0669","year":2013},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.488356Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:5c6bbd59e91a83c68fa0f7f5698f6a6d292edf08d301bff46e765f0064e0aac0","observation_id":"7e3f5be5-f298-455e-97e7-d651ab6e6707","resolution":{"observed_at":"2026-08-10T16:18:17.884630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.02141","last_updated":"2020-10-05T16:40:38Z","snapshot_observed_at":"2026-08-10T09:11:19.707395Z","submitted_at":"2020-10-05T16:40:38Z","title":"AdaLead: A simple and robust adaptive greedy search algorithm for sequence design","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.02141","snapshot_observed_at":"2026-08-10T16:18:17.491734Z","title":null,"venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.491734Z"},"links":{"cited_paper":"/paper/2010.02141","citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:ef6c8a309c7db0c02f0e3cb0e004b3edc1c5b57d5d9a37900204a8b9a0c0c2b8","observation_id":"0d8d3c58-3fb5-442a-910e-1fbcf410a061","resolution":{"observed_at":"2026-08-10T16:18:17.491734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.871137Z","title":null,"venue":null,"work_id":"f84db4a7-3346-4e61-8282-606cd1f4887f","year":2008},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.495212Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:f64d4676b72ecc211f3935cd6d712b765c1d57c83834a59267a6722d5845ecb7","observation_id":"7b349bac-5711-47df-8239-923813f60ffc","resolution":{"observed_at":"2026-08-10T16:18:17.874515Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.498464Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.498464Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:85d6b166b4850f0a795b8010c7aed69cf18a3d75f177e9fd88c16dedd20f8bac","observation_id":"d28d934f-8511-4603-ae20-c3589b6c9f0e","resolution":{"observed_at":"2026-08-10T16:18:17.498464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.855674Z","title":"and Littman, M","venue":null,"work_id":"7d51a143-9b5b-4b82-b92d-affaa0bf1592","year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.501817Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:e37ce054ec21da328eb6f9b80d7bbf7d95d21926a072f4f362e45c793dc07d88","observation_id":"7b9ddedf-1641-4e2a-9273-88abf6a72103","resolution":{"observed_at":"2026-08-10T16:18:17.858990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.845803Z","title":"A., Courville, A., and Bellemare, M","venue":null,"work_id":"8089ab40-3cdf-4f60-8236-780b5dd32dbd","year":2022},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.505364Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:d9388fddd65d9e6f7f52a5c21a668d4f474ade6667e7174832774110bc837e71","observation_id":"bb8f331d-fd14-4160-a974-7e47818eecd8","resolution":{"observed_at":"2026-08-10T16:18:17.849124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.835720Z","title":"Performance loss bounds for approximate value iteration with state aggregation","venue":null,"work_id":"916added-f27f-4dd4-98c9-b694e382277d","year":2006},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.508934Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:fbaf7605ec5a2f6f3e5860e86262740521d5b218a2f68cd6b1e2644c7ff4a94b","observation_id":"05d7cf58-d422-40d2-bcf9-3bcfdc45efb3","resolution":{"observed_at":"2026-08-10T16:18:17.839415Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.824667Z","title":"A concise introduction to multiagent systems and distributed artificial intelligence","venue":null,"work_id":"880fdd79-fbc0-4484-9d58-bdfe38e7e6d5","year":2022},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.512242Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:63e4446d9879cb2c917302a129389c22e5a062b125f978ea8a26f3582c8f80a0","observation_id":"0b83ea6f-9e4b-46ea-b154-a90b8b0709e0","resolution":{"observed_at":"2026-08-10T16:18:17.828359Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.815211Z","title":"and Klabjan, D","venue":null,"work_id":"130a9852-877a-40e1-a50e-79ed871b0e8c","year":2018},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.515848Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:03fea4064e620ee3384a552036340d6df36766017329f085b92958e085395787","observation_id":"afccec48-a659-4ca5-bd62-4b9d2bd4ca44","resolution":{"observed_at":"2026-08-10T16:18:17.818329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.804364Z","title":"Multiagent systems: a modern approach to distributed artificial intelligence","venue":null,"work_id":"c30e140a-7831-4d9b-b1b5-553bf8c64568","year":1999},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.519228Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:63698ddbf77ad5ff232c107d67852c55e0b8fce50a6f79234c28bb20771e5a2b","observation_id":"c062aba4-ce0e-438d-aaf1-960917a743f1","resolution":{"observed_at":"2026-08-10T16:18:17.807833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.794180Z","title":"and Van Roy, B","venue":null,"work_id":"cc8bbb01-70ba-40d8-a63d-375ba501ab8b","year":2017},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.522440Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:7bda8a26724e159d0d58b3390c978b7dfa3ca848c671963e436bf2f99168e6f5","observation_id":"713f4f87-9a06-46a1-ad8e-3898b70d69c1","resolution":{"observed_at":"2026-08-10T16:18:17.797372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2211.15931","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.628544Z","title":"Posterior sampling for continuing environments","venue":null,"work_id":"05a65ced-c423-4f49-8fae-3e97c0b52521","year":2022},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.525638Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:0d16ed40a6d6223f7446755facf489e6cc3d886ca7469c2d1d19f093fc673c9b","observation_id":"6da1f634-823e-4393-b905-5da185e79c4b","resolution":{"observed_at":"2026-08-10T16:18:17.635551Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.784742Z","title":"Frequentist regret bounds for randomized least-squares value iteration","venue":null,"work_id":"3d53e46e-6254-4ed2-af6b-fd18fdde0887","year":1954},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.528886Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:a762e19104fcaab023ab064635ea088c1d7274b52a28e5b440b45360ee3f4e98","observation_id":"14a15f90-cae8-45d7-b3a3-688afba633b8","resolution":{"observed_at":"2026-08-10T16:18:17.788059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.532317Z","title":"Multi-agent reinforcement learning: A selective overview of theories and algorithms","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.532317Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:c7e70a141bfa9183e9ed43e877bc53fb5ddcc29376572f5319547f6d40298901","observation_id":"1d45f6f4-e241-4dca-8371-45ce7b77c17c","resolution":{"observed_at":"2026-08-10T16:18:17.532317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:18:17.769258Z","title":"Multi-agent cooperative reinforcement learning in 3d virtual world","venue":null,"work_id":"3afc138b-5a6d-4d53-ac93-08b5bba1ee89","year":2010},"citing_paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration","version":3},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-10T16:18:17.535895Z"},"links":{"citing_paper":"/paper/2501.13394"},"observation_digest":"sha256:2bc73d32731296db1e09852b5e409cf51c9c71558076c30fa6537fc4c0355051","observation_id":"628b2879-55cb-4c71-90eb-6e91781539f8","resolution":{"observed_at":"2026-08-10T16:18:17.772738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.13394","last_updated":"2025-06-15T19:49:41Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-11T05:35:29.939153Z","submitted_at":"2025-01-23T05:37:33Z","title":"Concurrent Learning with Aggregated States via Randomized Least Squares Value Iteration"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":5,"verified_fuzzy":34},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2501.13394."}