{"as_of":"2026-08-12T12:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:53cac0942906ce0ecee4f1428174476e33dff2d9373c477fcb35228d76904dc2","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":22,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":22,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":22,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T04:54:49.069229Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T14:09:52.701191Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"1906.09237","last_updated":"2019-06-24T15:36:55Z","snapshot_observed_at":"2026-07-06T08:01:59.671409Z","submitted_at":"2019-06-21T16:54:42Z","title":"Shaping Belief States with Generative Environment Models for RL","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-25T18:56:19.459447Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/1906.09237"},"observation_digest":"sha256:6c4ec1955049989eff938e8053b0591c73102914b0b72911c7f99e886daf35cb","observation_id":"01cb16bc-ee80-47b5-8119-b8eecee0f75e","resolution":{"observed_at":"2026-05-25T18:57:06.758886Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"1906.12266","last_updated":"2019-06-28T15:35:11Z","snapshot_observed_at":"2026-07-06T08:03:38.968054Z","submitted_at":"2019-06-28T15:35:11Z","title":"Growing Action Spaces","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-25T13:24:59.682609Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/1906.12266"},"observation_digest":"sha256:e3eeaf911460f5fee2652745dc8f1d994bf688d74de8af345db527d19f6d6c97","observation_id":"8d9172cd-bbb9-4d08-acab-e7bd88f61f5e","resolution":{"observed_at":"2026-05-25T13:25:52.358606Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"1907.05300","last_updated":"2019-07-11T15:20:50Z","snapshot_observed_at":"2026-08-05T17:04:59.926732Z","submitted_at":"2019-07-11T15:20:50Z","title":"Learning Safe Unlabeled Multi-Robot Planning with Motion Constraints","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-24T23:14:06.926188Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/1907.05300"},"observation_digest":"sha256:11bee9a12b7150b16ac85051bbaf0c9580b950bba70a52747d21e6f6a97faf03","observation_id":"e6691caa-35d9-48f6-a089-98eb4e6a0373","resolution":{"observed_at":"2026-05-24T23:15:03.078689Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"1912.01603","last_updated":"2020-03-17T17:10:58Z","snapshot_observed_at":"2026-08-09T02:21:17.479736Z","submitted_at":"2019-12-03T18:57:16Z","title":"Dream to Control: Learning Behaviors by Latent Imagination","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T01:16:36.399272Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/1912.01603"},"observation_digest":"sha256:b4db683b3f73ea2f8bc8402cf7bc63ef141dcf3822ee43fc54b5f7d505c3ca59","observation_id":"93d6e0d7-947c-457a-a553-f5a87b26683d","resolution":{"observed_at":"2026-05-12T01:16:36.481403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"1912.06680","last_updated":"2019-12-13T19:56:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-12-13T19:56:40Z","title":"Dota 2 with Large Scale Deep Reinforcement Learning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-12T22:18:15.591361Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/1912.06680"},"observation_digest":"sha256:c3f5444574dd4af2a9ec42445f29c37ffdfab47e0057b9a90d7f4885320dc0ee","observation_id":"c641b33c-8e7f-405f-acf7-910c8374e3f7","resolution":{"observed_at":"2026-05-12T22:18:15.786172Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2005.01643","last_updated":"2020-11-01T23:50:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-04T17:00:15Z","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","version":3},"reference_index":295,"source":"arxiv_source","source_observed_at":"2026-05-11T11:33:20.892688Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2005.01643"},"observation_digest":"sha256:7297e0e477e1fa065c89b6e1a359184e27fc8e5bcefa1a6202ecdc8aeb3110b4","observation_id":"01b67e85-c8f9-4bd1-bfe4-702ab7aaa0ce","resolution":{"observed_at":"2026-05-11T11:33:21.664988Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2301.04104","last_updated":"2024-04-17T17:41:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-10T18:12:16Z","title":"Mastering Diverse Domains through World Models","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-11T09:08:21.677362Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2301.04104"},"observation_digest":"sha256:22d9be2ac48862c5c459c9393e74e42f4bfa31620044c5c4c03002b1fd7b5e7c","observation_id":"6620b507-68df-41b9-95ea-d3089bcdcfce","resolution":{"observed_at":"2026-05-11T09:08:21.921972Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-12T04:54:49.069229Z","title":"Impala: Scalable distributed deep-rl with importance weighted actor-learner architectures, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.00944","last_updated":"2024-12-01T19:32:04Z","snapshot_observed_at":"2026-08-12T04:48:25.357284Z","submitted_at":"2024-12-01T19:32:04Z","title":"Bilinear Convolution Decomposition for Causal RL Interpretability","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-12T04:54:49.069229Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2412.00944"},"observation_digest":"sha256:5861c28f1f3ade0969a0624fdad64cb49f65f2def81527e679dc656ee8dcc474","observation_id":"30932dcd-0afc-4cf7-b23d-1156aedd7cc6","resolution":{"observed_at":"2026-08-12T04:54:49.069229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-11T16:48:42.955351Z","title":"(02 2018)","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.09814","last_updated":"2025-02-05T19:35:48Z","snapshot_observed_at":"2026-08-11T16:39:41.643018Z","submitted_at":"2024-12-13T03:09:35Z","title":"Federated Learning of Dynamic Bayesian Network via Continuous Optimization from Time Series Data","version":2},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-11T16:48:42.955351Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2412.09814"},"observation_digest":"sha256:17f7e20f586d974d42fd9e07fee7e2476ec30e21e8effb3ff7d037b1b0d8f3e0","observation_id":"8e9b4180-c2af-4ab1-9420-7e91d9c3ec09","resolution":{"observed_at":"2026-08-11T16:48:42.955351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-10T14:46:40.896067Z","title":"Espeholt, H","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.15034","last_updated":"2025-01-25T02:35:46Z","snapshot_observed_at":"2026-08-12T05:45:42.839841Z","submitted_at":"2025-01-25T02:35:46Z","title":"Divergence-Augmented Policy Optimization","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-10T14:46:40.896067Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2501.15034"},"observation_digest":"sha256:521efef1ea06548111bdc0f783774daaadb74e78f13b663247c8af384d4d4953","observation_id":"6467b71b-4615-43d1-9bfb-192d6363263d","resolution":{"observed_at":"2026-08-10T14:46:40.896067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-07T15:24:07.347431Z","title":"IMPALA: Scal- able Distributed Deep-RL with Importance Weighted Actor- Learner Architectures,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15863","last_updated":"2025-05-21T07:59:18Z","snapshot_observed_at":"2026-08-07T23:22:27.322342Z","submitted_at":"2025-05-21T07:59:18Z","title":"Generative AI for Autonomous Driving: A Review","version":1},"reference_index":257,"source":"pdf_text","source_observed_at":"2026-08-07T15:24:07.347431Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2505.15863"},"observation_digest":"sha256:400b9086a89367d2f7440a32a3c1cf76d60a03fda3f259e3f6cacf6878cb515c","observation_id":"03d0c68a-fe9d-415c-a8ea-b0196c1bf0c6","resolution":{"observed_at":"2026-08-07T15:24:07.347431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-07T14:47:31.462520Z","title":"Impala: Scalable distributed deep-rl with importance weighted actor-learner architectures, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.17673","last_updated":"2025-05-23T09:38:55Z","snapshot_observed_at":"2026-08-10T03:59:50.427391Z","submitted_at":"2025-05-23T09:38:55Z","title":"Rethinking Agent Design: From Top-Down Workflows to Bottom-Up Skill Evolution","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:31.462520Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2505.17673"},"observation_digest":"sha256:f3ea11358755fcc61d4c1bcbbcb2f6606ff37f4cd2b5ff28bfbffad3a05e57dd","observation_id":"0a90ddc6-33be-42a8-bfe5-3425ed2a759d","resolution":{"observed_at":"2026-08-07T14:47:31.462520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-07T04:42:27.819005Z","title":"https://distill.pub/2019/activation-atlas","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.10138","last_updated":"2026-05-26T19:10:13Z","snapshot_observed_at":"2026-08-09T21:45:32.789653Z","submitted_at":"2025-06-11T19:36:17Z","title":"Path Channels and Plan Extension Kernels: a Mechanistic Description of Planning in a Sokoban RNN","version":3},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-07T04:42:27.819005Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2506.10138"},"observation_digest":"sha256:658bf27c19e8ff1e66fcf972c0904109b3969967b3163f5de2ad0b81078c240f","observation_id":"fac34978-7ca8-41c5-85b1-5523b877bc77","resolution":{"observed_at":"2026-08-07T04:42:27.819005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-06T22:23:50.021754Z","title":"Impala: Scalable distributed deep-rl with importance weighted actor-learner architectures,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.21782","last_updated":"2025-06-26T21:39:01Z","snapshot_observed_at":"2026-08-11T04:55:16.534512Z","submitted_at":"2025-06-26T21:39:01Z","title":"M3PO: Massively Multi-Task Model-Based Policy Optimization","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:23:50.021754Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2506.21782"},"observation_digest":"sha256:f3813433de43e583db7b40a3f8fdaeb84b82e13bd032abfe364b1e41d78444a2","observation_id":"24d7b1a2-8640-4d15-aa2a-d20528885b22","resolution":{"observed_at":"2026-08-06T22:23:50.021754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2605.03065","last_updated":"2026-06-26T00:04:01Z","snapshot_observed_at":"2026-08-11T23:39:50.755664Z","submitted_at":"2026-05-04T18:36:40Z","title":"OGPO: Sample Efficient Full-Finetuning of Generative Control Policies","version":1},"reference_index":142,"source":"arxiv_source","source_observed_at":"2026-05-08T18:48:56.075160Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2605.03065"},"observation_digest":"sha256:c29d16e456c2b3cfb89c234d0e775b4ffc7929e8b31f6015e383d9cdc586c7c9","observation_id":"a8e162b0-6260-4029-9169-179e10bf90f1","resolution":{"observed_at":"2026-05-09T06:10:42.435634Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2605.03065","last_updated":"2026-06-26T00:04:01Z","snapshot_observed_at":"2026-08-11T23:39:50.755664Z","submitted_at":"2026-05-04T18:36:40Z","title":"OGPO: Sample Efficient Full-Finetuning of Generative Control Policies","version":4},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-07-01T00:02:55.449923Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2605.03065"},"observation_digest":"sha256:701d5560fe9cbc869eabee5c29ba39eedb055bc4a7a7eb3d162261741dc12aac","observation_id":"c400599a-47be-4c03-b5fc-8696aad2b151","resolution":{"observed_at":"2026-07-01T00:05:09.721995Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2605.05481","last_updated":"2026-06-17T20:11:22Z","snapshot_observed_at":"2026-08-11T20:41:31.375474Z","submitted_at":"2026-05-06T22:02:35Z","title":"Approximate Next Policy Sampling: Replacing Conservative Target Policy Updates in Deep RL","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-08T16:58:39.520620Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2605.05481"},"observation_digest":"sha256:3d3783d28af449b2c15bcf31cf7ef54a8fc1c83b10f168aa06f260c8bae913b5","observation_id":"f0b3a030-5ac9-4eed-9226-7b1e879ead9e","resolution":{"observed_at":"2026-05-11T17:56:05.312851Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2605.05481","last_updated":"2026-06-17T20:11:22Z","snapshot_observed_at":"2026-08-11T20:41:31.375474Z","submitted_at":"2026-05-06T22:02:35Z","title":"Approximate Next Policy Sampling: Replacing Conservative Target Policy Updates in Deep RL","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T23:41:30.860309Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2605.05481"},"observation_digest":"sha256:a13340f0d449db1be26b13039bde4eec9d4053771c8ad8d55d6655b32e3bc0f1","observation_id":"67e90f2d-fd7c-4ba9-8993-fba9c2b4e2b9","resolution":{"observed_at":"2026-06-30T23:45:08.134398Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2606.08446","last_updated":"2026-06-07T04:24:45Z","snapshot_observed_at":"2026-08-05T08:26:57.919131Z","submitted_at":"2026-06-07T04:24:45Z","title":"Sparrow: Sparse Rollout for Stable and Efficient Long-context RL of Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T19:10:53.882876Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2606.08446"},"observation_digest":"sha256:4aa60edc6331834e471a4556aa44d0a999018be4f21c179698adb02bd437dc39","observation_id":"795112a6-2e44-45b6-a3e3-63ce5fd9bae1","resolution":{"observed_at":"2026-07-02T22:07:26.508407Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2606.27348","last_updated":"2026-06-25T17:54:36Z","snapshot_observed_at":"2026-07-07T00:01:34.696890Z","submitted_at":"2026-06-25T17:54:36Z","title":"Bridging Performance and Generalization in Reinforcement Learning for Agile Flight","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-26T04:33:20.981811Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2606.27348"},"observation_digest":"sha256:25a6ff64bd2ed2a0315af319748115f13c175674520dc41a5c1c8d6ea0465017","observation_id":"e784af15-4b08-43b3-987b-f5840a9852cf","resolution":{"observed_at":"2026-07-04T14:09:52.702576Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":"1802.01561","doi":null,"metadata_source":"pith","pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-07-04T14:09:52.701191Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","venue":"cs.LG","work_id":"9bbfa31b-454d-4174-923d-06e74e0c6cab","year":2018},"citing_paper":{"arxiv_id":"2606.27580","last_updated":"2026-06-25T22:12:22Z","snapshot_observed_at":"2026-08-02T08:38:09.113444Z","submitted_at":"2026-06-25T22:12:22Z","title":"Retroactive Advantage Correction: Closed-Form V-Trace Bias Correction for Delay-Aware RLHF","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T01:23:07.441218Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2606.27580"},"observation_digest":"sha256:d1d01d6b0f50c372b45dc8ebe404fce522a0ab5007c0bf87420ac27fcc165f39","observation_id":"5f4ac08a-1039-44ae-9c64-b296258edbab","resolution":{"observed_at":"2026-07-01T18:55:58.783845Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.01561","snapshot_observed_at":"2026-08-01T07:56:35.435467Z","title":"Impala: Scalable distributed deep-rl with importance weighted actor-learner architectures,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.21290","last_updated":"2026-07-23T13:12:51Z","snapshot_observed_at":"2026-08-09T05:51:16.024724Z","submitted_at":"2026-07-23T13:12:51Z","title":"Multi-Task Learning for Heterogeneous Prediction from Video Game State with Transfer Learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T07:56:35.435467Z"},"links":{"cited_paper":"/paper/1802.01561","citing_paper":"/paper/2607.21290"},"observation_digest":"sha256:9738a54ccb38c4d4c28c1bd0dabf8683d15e8021daa1e8cbd5a1c2aa999da55c","observation_id":"0374c552-206a-49c1-b20c-2af14c8ba37e","resolution":{"observed_at":"2026-08-01T07:56:35.435467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1802.01561/citation-record","integrity":"/paper/1802.01561/integrity","json":"/paper/1802.01561/citation-record.json","paper":"/paper/1802.01561"},"outbound":[],"paper":{"arxiv_id":"1802.01561","last_updated":"2018-06-28T06:54:39Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T06:21:49.379500Z","submitted_at":"2018-02-05T18:47:30Z","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 22 inbound Pith citation observations for arXiv:1802.01561."}