{"as_of":"2026-08-18T01:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3c2d4cd349cd8c34d7d9ab5de966ba4ea0a6951fc9a5b50fe59c9bf48985f606","coverage":[{"denominator":16,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-21T21:33:40.229376Z","state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T21:42:30.436671Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.02590","snapshot_observed_at":"2026-08-02T21:42:30.436671Z","title":"Gelada, C., Kumar, S., Buckman, J., Nachum, O., and Belle- mare, M","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.19373","last_updated":"2026-06-04T04:14:07Z","snapshot_observed_at":"2026-08-13T19:32:21.863609Z","submitted_at":"2026-02-22T22:55:27Z","title":"Stable Deep Reinforcement Learning via Isotropic Gaussian Representations","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T21:42:30.436671Z"},"links":{"cited_paper":"/paper/2510.02590","citing_paper":"/paper/2602.19373"},"observation_digest":"sha256:f595bfaabc7933adee57472e252458f3578e8b302c6c085e55195f656a2cabac","observation_id":"81145edf-9c8f-4fdb-b245-a69fa66192cc","resolution":{"observed_at":"2026-08-02T21:42:30.436671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2510.02590/citation-record","integrity":"/paper/2510.02590/integrity","json":"/paper/2510.02590/citation-record.json","paper":"/paper/2510.02590"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1812.06110","last_updated":"2018-12-14T19:03:38Z","snapshot_observed_at":"2026-08-14T19:03:45.460809Z","submitted_at":"2018-12-14T19:03:38Z","title":"Dopamine: A Research Framework for Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"1812.06110","doi":null,"metadata_source":"pith","pith_arxiv_id":"1812.06110","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dopamine: A Research Framework for Deep Reinforcement Learning","venue":"cs.LG","work_id":"a0c1e29f-e775-49dc-977e-bbbca0daa6d8","year":2018},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1812.06110","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:85c083d7e17370ab02ed3aa1548592305fdd577c2597f65526469f1196137de0","observation_id":"d57b61d9-ac4c-46e7-88e6-00379eac8373","resolution":{"observed_at":"2026-05-21T21:34:22.294809Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.05905","last_updated":"2019-01-29T12:10:47Z","snapshot_observed_at":"2026-08-17T07:51:41.384508Z","submitted_at":"2018-12-13T04:44:29Z","title":"Soft Actor-Critic Algorithms and Applications","version":2},"cited_work":{"arxiv_id":"1812.05905","doi":"10.48550/arxiv.1812.05905","metadata_source":"pith","pith_arxiv_id":"1812.05905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Soft Actor-Critic Algorithms and Applications","venue":"cs.LG","work_id":"bb49c9fb-03b2-4226-9edb-50186b8193e4","year":2018},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1812.05905","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:c0fa922a165dc307d022a4a09a86de9853a158b0190e08ffae5f78ad5dcd1673","observation_id":"f473e237-c902-4191-878e-2294cefac0a7","resolution":{"observed_at":"2026-05-21T21:34:22.323552Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-10T21:38:16.985704+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-10T21:38:16.985704+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15280","last_updated":"2025-05-29T14:58:32Z","snapshot_observed_at":"2026-08-16T12:56:25.698544Z","submitted_at":"2025-02-21T08:17:24Z","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2502.15280","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15280","snapshot_observed_at":"2026-07-04T20:30:07.802722Z","title":"Hyperspher- ical normalization for scalable deep reinforcement learning.arXiv preprint arXiv:2502.15280","venue":null,"work_id":"8def73dd-bca2-4abe-a959-f2bfe11ad210","year":2025},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/2502.15280","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:7775b1694eacff8629930576064bf57eebbcb587b9b6df906f3af03b658c1283","observation_id":"a15785bd-3cd3-434a-b66b-5b95ea1f79b5","resolution":{"observed_at":"2026-05-21T21:34:22.302146Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18719","last_updated":"2025-05-24T14:42:51Z","snapshot_observed_at":"2026-08-14T01:02:35.686274Z","submitted_at":"2025-05-24T14:42:51Z","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2505.18719","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.18719","snapshot_observed_at":"2026-07-10T23:07:47.932038Z","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","venue":"cs.RO","work_id":"7bc1dc16-0cc4-4159-8dd5-180b59579c5e","year":2025},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/2505.18719","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:6e5f561d4f86ee55e575fe12341d0403f7f7f5c30115fd8f307031b3aaab4c9f","observation_id":"f4cc08e2-a87f-432b-ab5f-013e68fd6365","resolution":{"observed_at":"2026-05-21T21:34:22.289168Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.5602","last_updated":"2013-12-19T16:00:08Z","snapshot_observed_at":"2026-08-17T17:19:33.060917Z","submitted_at":"2013-12-19T16:00:08Z","title":"Playing Atari with Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"1312.5602","doi":"10.48550/arxiv.1312.5602","metadata_source":"pith","pith_arxiv_id":"1312.5602","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Playing Atari with Deep Reinforcement Learning","venue":"cs.LG","work_id":"736a8ddf-e365-4940-ad58-4699fddedb86","year":2013},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1312.5602","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:b83d09af64cfddb6026fe0553e115d75fe672053b30a6c2a17887a6a4b24f7cc","observation_id":"b16a3702-e63b-4c3e-9d86-62f537a9605a","resolution":{"observed_at":"2026-05-21T21:34:22.290821Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-08-15T20:26:32.102285Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:0fa41cde966a8dfb3a7f3bb943e1ac8b087dd1283ba9c3a89bd1806115214e6d","observation_id":"9573638c-3916-440d-b4bd-c68c2c3a6b68","resolution":{"observed_at":"2026-05-21T21:34:22.318049Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10506","last_updated":"2024-06-18T18:11:07Z","snapshot_observed_at":"2026-08-16T14:09:21.685692Z","submitted_at":"2024-03-15T17:45:44Z","title":"HumanoidBench: Simulated Humanoid Benchmark for Whole-Body Locomotion and Manipulation","version":2},"cited_work":{"arxiv_id":"2403.10506","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.10506","snapshot_observed_at":"2026-07-04T19:40:06.916232Z","title":"Humanoid- bench: Simulated humanoid benchmark for whole-body locomotion and manipulation.arXiv preprint arXiv:2403.10506","venue":null,"work_id":"8e42d104-ddc6-4291-b477-f09257aa5c80","year":2024},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/2403.10506","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:5257d5b2da6e038e6dc63a8cf9e1b682429d423c71e26a6fc948ee8375532ef2","observation_id":"b41db84b-b6d6-4ed8-9a55-daf50f968a25","resolution":{"observed_at":"2026-05-21T21:34:22.312178Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-14T20:06:37.819179Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:89cde611b27b17f02682577698c6a1752848991c60db1149bb2ba95475cbd221","observation_id":"72b104ca-c66f-4952-b4b4-a77ff7925207","resolution":{"observed_at":"2026-05-21T21:34:22.300517Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1801.00690","last_updated":"2018-01-02T15:48:14Z","snapshot_observed_at":"2026-08-01T20:24:08.300098Z","submitted_at":"2018-01-02T15:48:14Z","title":"DeepMind Control Suite","version":1},"cited_work":{"arxiv_id":"1801.00690","doi":"10.48550/arxiv.1801.00690","metadata_source":"pith","pith_arxiv_id":"1801.00690","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepMind Control Suite","venue":"cs.AI","work_id":"54294ef0-c651-4d5a-a72b-f85a88329a71","year":2018},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1801.00690","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:62afb4b25f186dde20ad4a443b6b34cd681c0c77446ecd650bcf624908219fe8","observation_id":"f64fb178-d907-4734-a8d3-54012a4eafb9","resolution":{"observed_at":"2026-05-21T21:34:22.305967Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.02648","last_updated":"2018-12-06T16:36:20Z","snapshot_observed_at":"2026-08-14T17:47:25.281874Z","submitted_at":"2018-12-06T16:36:20Z","title":"Deep Reinforcement Learning and the Deadly Triad","version":1},"cited_work":{"arxiv_id":"1812.02648","doi":"10.48550/arxiv.1812.02648","metadata_source":"pith","pith_arxiv_id":"1812.02648","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deep Reinforcement Learning and the Deadly Triad","venue":"cs.AI","work_id":"de214ead-4cb0-4abd-be3d-ae3389f55e9b","year":2018},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"cited_paper":"/paper/1812.02648","citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:8b541ea9754b1da9fb6d64e30fcef7dac277de0c7536c4f19079d51970e2c1e0","observation_id":"cd31e76f-1ba0-4de1-84ba-6676815879c5","resolution":{"observed_at":"2026-05-21T21:34:22.282341Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Under Review","venue":null,"work_id":"6b8f3edc-d620-43d3-8425-54b2447781e8","year":2020},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:433e8cfb0c89c6e13a4ea35a4f8c834b737ffa808d0dd7fd07ea8abc0b3494c8","observation_id":"e68fcd9a-b01c-4b3e-b8b9-c747bbedeae8","resolution":{"observed_at":"2026-05-21T21:34:22.376244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The second inequality holds because theminoperator is also a non-expansion","venue":null,"work_id":"5d458e14-b694-43e9-a416-a342ec084530","year":2020},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:924b3525bd89521c8e947d2dd96381af51fc4d647de5db68ec792d213e767d3d","observation_id":"43ab906f-195e-4aae-af77-35b800c8c918","resolution":{"observed_at":"2026-05-21T21:34:22.380690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"4, SAC is adapted to use a single Q-function critic, following the approach taken in Simba (Lee et al","venue":null,"work_id":"6883074c-40b1-4f8e-8ab7-f3d31e2eb03f","year":2024},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:1739917e445327f5a8cf7b842d9116a2942fc567cabff076e431b80e837b3149","observation_id":"a23c9dd6-70b0-4424-90ca-64f62a555cb1","resolution":{"observed_at":"2026-05-21T21:34:22.372047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Under Review","venue":null,"work_id":"7a5a9330-9da9-4220-ae04-67a9ae8ad52d","year":2020},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:e43b588766bc2ff44a5e9cbe7364aab81b0c58923476607f9995101d2eab43b8","observation_id":"189f580a-e9d7-49db-a419-a87d2ee50837","resolution":{"observed_at":"2026-05-21T21:34:22.392002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"C.4 ONLINERLANDCONTINUOUSCONTROL For our continuous-control experiments with online reinforcement learning, we adopt SimbaV1 and SimbaV2","venue":null,"work_id":"b470e532-ce6c-4fe6-bc55-7f5cf96d3194","year":2000},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:afbb9b8ab1d66511d24b998f028d219924b490decf66caaf62016436d55262b6","observation_id":"c1a29729-c6ac-49a7-bafd-4007fc18ad7b","resolution":{"observed_at":"2026-05-21T21:34:22.384735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Identical values are merged","venue":null,"work_id":"b9974ada-e53e-4216-910d-eb9a574c2a77","year":2000},"citing_paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T21:33:40.229376Z"},"links":{"citing_paper":"/paper/2510.02590"},"observation_digest":"sha256:85e9851e36fa2b9433cacfe720fcbd281334c7c91ae3f031c3b65654accfaf08","observation_id":"a3a20ee3-2558-4d09-ab56-ae8a34af91af","resolution":{"observed_at":"2026-05-21T21:34:22.388263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2510.02590","last_updated":"2026-05-18T11:40:07Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-16T16:50:40.382915Z","submitted_at":"2025-10-02T21:48:01Z","title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning"},"reference_resolution":{"displayed":16,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":9,"verified_fuzzy":6},"total_outbound_references":16},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 16 of 16 outbound references and 1 inbound Pith citation observation for arXiv:2510.02590."}