{"as_of":"2026-08-12T16:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:807f0d57f2ce9dbd449fc92f4ddf1d3b94c33b95bf456a9e567581d9b2a1af00","coverage":[{"denominator":112,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T12:27:15.062636Z","state":"measured"},{"denominator":104,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":104,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T00:07:41.420435Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T13:29:51.963553Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14355","snapshot_observed_at":"2026-08-11T00:07:41.420435Z","title":"Enabling realtime reinforcement learning at scale with staggered asynchronous inference, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19726","last_updated":"2025-06-12T14:58:31Z","snapshot_observed_at":"2026-08-10T23:53:24.513179Z","submitted_at":"2024-12-27T16:30:12Z","title":"Position: Theory of Mind Benchmarks are Broken for Large Language Models","version":4},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-11T00:07:41.420435Z"},"links":{"cited_paper":"/paper/2412.14355","citing_paper":"/paper/2412.19726"},"observation_digest":"sha256:6b8e3640fe0f60ccc0b8fd1ff7d31fde3226146677f73b02887c6830fed23c18","observation_id":"de804997-082a-43ea-92b9-a3953a2a84a2","resolution":{"observed_at":"2026-08-11T00:07:41.420435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14355","snapshot_observed_at":"2026-08-04T23:17:52.993979Z","title":"Enabling realtime reinforcement learning at scale with staggered asynchronous inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.06714","last_updated":"2025-09-08T14:09:33Z","snapshot_observed_at":"2026-08-09T19:35:38.686867Z","submitted_at":"2025-09-08T14:09:33Z","title":"RT-HCP: Dealing with Inference Delays and Sample Efficiency to Learn Directly on Robotic Platforms","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T23:17:52.993979Z"},"links":{"cited_paper":"/paper/2412.14355","citing_paper":"/paper/2509.06714"},"observation_digest":"sha256:a77a6d6634724a84d83c6cc95271bcb73774d6938b5b5514bc17b3daa6d1c21f","observation_id":"7a2b46ba-fbf4-420c-bf80-95161b1c80c2","resolution":{"observed_at":"2026-08-04T23:17:52.993979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"cited_work":{"arxiv_id":"2412.14355","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.14355","snapshot_observed_at":"2026-07-04T13:29:51.963553Z","title":"Under review","venue":null,"work_id":"f1d3f7ae-14ca-4854-8e4d-2007cf71b528","year":null},"citing_paper":{"arxiv_id":"2606.26463","last_updated":"2026-06-27T05:05:34Z","snapshot_observed_at":"2026-08-03T01:41:10.487866Z","submitted_at":"2026-06-24T23:55:15Z","title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-26T05:08:19.504454Z"},"links":{"cited_paper":"/paper/2412.14355","citing_paper":"/paper/2606.26463"},"observation_digest":"sha256:c7cf26597120701aa69147c9ac99de19d986bdfc794038a75d4c503aec5f6bc3","observation_id":"6e89106f-9536-49e9-9b41-6d31c9ab3a81","resolution":{"observed_at":"2026-07-04T13:29:51.964800Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"cited_work":{"arxiv_id":"2412.14355","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.14355","snapshot_observed_at":"2026-07-04T13:29:51.963553Z","title":"Under review","venue":null,"work_id":"f1d3f7ae-14ca-4854-8e4d-2007cf71b528","year":null},"citing_paper":{"arxiv_id":"2606.26463","last_updated":"2026-06-27T05:05:34Z","snapshot_observed_at":"2026-08-03T01:41:10.487866Z","submitted_at":"2026-06-24T23:55:15Z","title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-30T09:26:31.944405Z"},"links":{"cited_paper":"/paper/2412.14355","citing_paper":"/paper/2606.26463"},"observation_digest":"sha256:2ddda9c3c7b07df430e990c22f65b98912931ff28c84ae74b960e0c56f563290","observation_id":"20e97d7f-8c81-448b-9e81-6fec7a9ca9ef","resolution":{"observed_at":"2026-06-30T09:34:34.896004Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.14355/citation-record","integrity":"/paper/2412.14355/integrity","json":"/paper/2412.14355/citation-record.json","paper":"/paper/2412.14355"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2109.09876","last_updated":"2022-04-24T01:20:42Z","snapshot_observed_at":"2026-08-11T19:05:09.381933Z","submitted_at":"2021-09-20T22:50:01Z","title":"Context-Specific Representation Abstraction for Deep Option Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.09876","snapshot_observed_at":"2026-08-11T12:27:14.639773Z","title":"Context-specific representation abstraction for deep option learning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.639773Z"},"links":{"cited_paper":"/paper/2109.09876","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:8f34f12d44657b2550f9dd251ca5705500255f588bffc9ab332568caaf49b8a9","observation_id":"42676e21-90fe-446a-b0fe-7caeef5aa0e3","resolution":{"observed_at":"2026-08-11T12:27:14.639773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.646038Z","title":"Blind decision making: Reinforcement learning with delayed observations","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.646038Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:f87933eeb1b48a36ce3f2d801163c6169a9476d0c5e70362fd2cf796643d4cb1","observation_id":"2d40583f-5efd-4973-89c4-2797ff8d7f54","resolution":{"observed_at":"2026-08-11T12:27:14.646038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.13242","last_updated":"2021-06-23T19:38:24Z","snapshot_observed_at":"2026-08-11T19:05:40.098294Z","submitted_at":"2020-04-28T02:13:12Z","title":"Efficient Black-Box Planning Using Macro-Actions with Focused Effects","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.13242","snapshot_observed_at":"2026-08-11T12:27:14.650868Z","title":"Efficient black-box planning using macro-actions with focused effects","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.650868Z"},"links":{"cited_paper":"/paper/2004.13242","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:49b36ed1343eb1efe29e8b6e9de47f58e503c93db92bf2f47e843416df80207e","observation_id":"c1e272b4-c239-47b7-a543-dac16b596e68","resolution":{"observed_at":"2026-08-11T12:27:14.650868Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.656009Z","title":"The option-critic architecture","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.656009Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e225062987e55b29fb866404064a5f738faf7b13cd608344fa6461830689ee36","observation_id":"71a818d4-11a9-4253-90f1-49fdedb815e8","resolution":{"observed_at":"2026-08-11T12:27:14.656009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.660540Z","title":"Reinforcement learning and its relationship to supervised learning","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.660540Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e71c0dc003a99c726b2241917b2e639927b854fb86cf97b8ca65abc3976526e4","observation_id":"cb1d5fbe-ffbb-4c57-8db9-278b9434b4f8","resolution":{"observed_at":"2026-08-11T12:27:14.660540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.09330","last_updated":"2019-04-19T20:21:23Z","snapshot_observed_at":"2026-08-09T16:47:02.591259Z","submitted_at":"2019-04-19T20:21:23Z","title":"Continual Learning with Self-Organizing Maps","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.09330","snapshot_observed_at":"2026-08-11T12:27:14.665152Z","title":"Continual learning with self-organizing maps","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.665152Z"},"links":{"cited_paper":"/paper/1904.09330","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:21870d83887744145d965dbba7094ad941b54bdde93d22e4e24fdfb55b15e8df","observation_id":"2f49b150-b874-4263-94a8-1a251ed295f8","resolution":{"observed_at":"2026-08-11T12:27:14.665152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.669925Z","title":"The arcade learning environment: An evaluation platform for general agents","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.669925Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:00550fe875972476b0f01fff6dc6f44f31ae92f099492613156c8a267cbd10ee","observation_id":"44fa17d6-3c07-4c5b-a62d-c1df64154080","resolution":{"observed_at":"2026-08-11T12:27:14.669925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.673979Z","title":"Dynamic programming and optimal control","venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.673979Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:35fe1750112efc3a133c1bffa96517a34ffd5892dde4e7af757881a3b9d4b321","observation_id":"194ed486-1d2b-4673-856e-269f0e3f2f01","resolution":{"observed_at":"2026-08-11T12:27:14.673979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.677966Z","title":"Reinforcement learning with random delays","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.677966Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:b866cea1998195663c07fbf16c9a6b84de7ab7e56af24a44f6460fd575b97323","observation_id":"169d906c-f333-4347-81e7-2436fe8d4bd7","resolution":{"observed_at":"2026-08-11T12:27:14.677966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-08-12T05:40:07.199512Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-08-11T12:27:14.682133Z","title":"Openai gym","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.682133Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:31088f0628b2c6a9bdc547f91697139d19042573ad6adee80b44a905371bbb99","observation_id":"9ca0c860-d6c6-4694-91d8-ab7882b8eee4","resolution":{"observed_at":"2026-08-11T12:27:14.682133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.686543Z","title":"Recursive routing networks: Learning to compose modules for language understanding","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.686543Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:4098e20cd5240008e668c4d4367ae9242dae0d93671ce04545425bdd3b2ab03e","observation_id":"94faf705-8ed9-472e-8412-03dfeaa07b3e","resolution":{"observed_at":"2026-08-11T12:27:14.686543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.690534Z","title":"Automatically composing representation transformations as a means for generalization","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.690534Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:1675499ad107ca7241f62d79366b3dd59545cd627c652f9344b8dd0e665f28d9","observation_id":"7fa4cb4a-60de-4f4e-aa8f-93d6caa884e9","resolution":{"observed_at":"2026-08-11T12:27:14.690534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.694381Z","title":"Leveraging procedural generation to benchmark reinforcement learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.694381Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:473b36022c99cdf2813bb043293ae82fae2547619f8f528f44a8d66267f805ad","observation_id":"e13ebc03-b3e9-4807-9a19-daf62942e8f0","resolution":{"observed_at":"2026-08-11T12:27:14.694381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.10374","last_updated":"2022-04-21T19:07:50Z","snapshot_observed_at":"2026-08-11T07:44:56.766067Z","submitted_at":"2022-04-21T19:07:50Z","title":"Learning how to Interact with a Complex Interface using Hierarchical Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.10374","snapshot_observed_at":"2026-08-11T12:27:14.698765Z","title":"Learning how to interact with a complex interface using hierarchical reinforcement learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.698765Z"},"links":{"cited_paper":"/paper/2204.10374","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:9d8a1ae0c680696dfe7cceb10f8f4fee980ab5ec08ffb9f4aa047902ceb67a4a","observation_id":"4d3f0895-871a-4fdf-b5bc-65912eb48c3f","resolution":{"observed_at":"2026-08-11T12:27:14.698765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.07743","last_updated":"2018-11-13T16:24:09Z","snapshot_observed_at":"2026-07-06T07:08:59.218192Z","submitted_at":"2018-10-17T19:19:36Z","title":"PepCVAE: Semi-Supervised Targeted Design of Antimicrobial Peptide Sequences","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.07743","snapshot_observed_at":"2026-08-11T12:27:14.703211Z","title":"Pepcvae: Semi-supervised targeted design of antimicrobial peptide sequences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.703211Z"},"links":{"cited_paper":"/paper/1810.07743","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:b4fb5655a4f4101c44cd9493fe262e4559cb913085f638caeaadc43bfb36f800","observation_id":"c5bb3772-46e2-4d74-a958-0bbc51ab575e","resolution":{"observed_at":"2026-08-11T12:27:14.703211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.707857Z","title":"Acting in delayed environments with non- stationary markov policies","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.707857Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:8ed521067e52493a307735524a3c0d749d38cc46a41a0f845a7ece28fbb38b4c","observation_id":"a5ab8989-f0ae-474b-b522-f621be23ba0f","resolution":{"observed_at":"2026-08-11T12:27:14.707857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.11992","last_updated":"2023-12-13T02:40:47Z","snapshot_observed_at":"2026-08-10T21:31:45.608588Z","submitted_at":"2021-01-28T13:35:37Z","title":"Acting in Delayed Environments with Non-Stationary Markov Policies","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.11992","snapshot_observed_at":"2026-08-11T12:27:14.712015Z","title":"Acting in delayed environments with non- stationary markov policies","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.712015Z"},"links":{"cited_paper":"/paper/2101.11992","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:fed6fde8c0bf00f64371df0778504c63de3a3639a7b7a8027358c7a080c006f8","observation_id":"13629a57-e29b-42a8-82e6-c075bb028faa","resolution":{"observed_at":"2026-08-11T12:27:14.712015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12805","last_updated":"2024-03-19T15:06:53Z","snapshot_observed_at":"2026-08-12T06:34:10.336349Z","submitted_at":"2024-03-19T15:06:53Z","title":"Contextual Moral Value Alignment Through Context-Based Aggregation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12805","snapshot_observed_at":"2026-08-11T12:27:14.716776Z","title":"Contextual moral value alignment through context-based aggregation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.716776Z"},"links":{"cited_paper":"/paper/2403.12805","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:a5389129374379d7afd7962c55c249e7138f2b88d9137e85915d09e630faba11","observation_id":"9a191e21-725c-4ed0-84e0-d800acdd5bb8","resolution":{"observed_at":"2026-08-11T12:27:14.716776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.12901","last_updated":"2019-04-29T18:40:15Z","snapshot_observed_at":"2026-07-06T07:49:17.886466Z","submitted_at":"2019-04-29T18:40:15Z","title":"Challenges of Real-World Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.12901","snapshot_observed_at":"2026-08-11T12:27:14.721266Z","title":"Challenges of real-world rein- forcement learning","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.721266Z"},"links":{"cited_paper":"/paper/1904.12901","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:7c54e2f26741bbad052b1724894804fcb94bdead355f0d9f3980dee630899c17","observation_id":"349b4bff-0547-47b2-b915-dff83405b97e","resolution":{"observed_at":"2026-08-11T12:27:14.721266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.725578Z","title":"Impala: Scalable distributed deep-rl with importance weighted actor-learner architectures","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.725578Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:51e866dd3b44cbc8d1dc76c07a82c6da4337308432ea82a44a8c63a19f29427a","observation_id":"de3df47f-eebc-44f7-91e5-a27c6e7581bf","resolution":{"observed_at":"2026-08-11T12:27:14.725578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.729690Z","title":"Reducing the cost of cycle-time tuning for real-world policy optimization","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.729690Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:1986cca8e991674991183928b667c87977da17cb92782a841c761b126476f8c4","observation_id":"5964f396-5f02-4494-bdd0-58bd445fade2","resolution":{"observed_at":"2026-08-11T12:27:14.729690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.733660Z","title":"Learning with opponent-learning awareness","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.733660Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:cb1b71dfaac4eb3ebe6a621d2247fba5ef5676a80d3716acfb98b88e1ae75acd","observation_id":"bd75baed-9b4d-4d18-bd29-8cdae67d2fd8","resolution":{"observed_at":"2026-08-11T12:27:14.733660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.737671Z","title":"Dice: The infinitely differentiable monte carlo estimator","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.737671Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:dd4ef5d7ee4bc5a30ecccf1f0caf00255824d36cccbd7f7ab8a5f657773c6004","observation_id":"ae080094-a2fb-47da-9889-ea18ad9ee9f4","resolution":{"observed_at":"2026-08-11T12:27:14.737671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.741631Z","title":"Harris, K","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.741631Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:dc47938995dd618b3cfc83213b756e41301a90646792e7b85cb21287bd448440","observation_id":"44b7e040-700e-49f0-b4d5-552a23d47862","resolution":{"observed_at":"2026-08-11T12:27:14.741631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.745545Z","title":"Alexandria: Extensible framework for rapid exploration of social media","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.745545Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:bb65d89da876aa6c5adf7762385feba9c75703af7cbf86076e3fede614deaa17","observation_id":"32d673fd-24fd-4d52-a0d1-eefa674a9b93","resolution":{"observed_at":"2026-08-11T12:27:14.745545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.750142Z","title":"Rainbow: Combining improvements in deep reinforcement learning","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.750142Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:4f902e8433d74509ff4404c5846f0358a85889b50c432bed348a37d5f195ca27","observation_id":"31f3d686-b03c-4046-b10b-e6d213177db5","resolution":{"observed_at":"2026-08-11T12:27:14.750142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.754482Z","title":"Texplore: real-time sample-efficient reinforcement learning for robots","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.754482Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:eb8407068cdf051cc85aef243eea41867b67ee07d70037d57268864398fa994d","observation_id":"e05a2e96-ad17-4189-9bbb-1a27e0d81f16","resolution":{"observed_at":"2026-08-11T12:27:14.754482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1105.1749","last_updated":"2011-05-21T14:15:28Z","snapshot_observed_at":"2026-08-10T08:54:54.727164Z","submitted_at":"2011-05-09T18:17:20Z","title":"A Real-Time Model-Based Reinforcement Learning Architecture for Robot Control","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1105.1749","snapshot_observed_at":"2026-08-11T12:27:14.758876Z","title":"A real-time model-based reinforcement learning architecture for robot control","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.758876Z"},"links":{"cited_paper":"/paper/1105.1749","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:7ea87c8af25e9b1c5460a6141df987c8764975860411bfdf86092334520eb005","observation_id":"df57cb1f-6ff2-4986-8d33-09c9f3979804","resolution":{"observed_at":"2026-08-11T12:27:14.758876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.763424Z","title":"Rtmba: A real-time model-based reinforce- ment learning architecture for robot control","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.763424Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:83d8894b763b04a230fc591b1e0bfd71ca0bb047d83dc8f9e149f2ef0391168b","observation_id":"ddfbd057-651a-47f1-8fd8-5b95952d4372","resolution":{"observed_at":"2026-08-11T12:27:14.763424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.767749Z","title":"Why can a machine beat mario but not pokemon?, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.767749Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:60540c5ea5f001557e2bb25248625ecd2e077640b8aec0b85637527c5f5d7dfd","observation_id":"7442c6fd-60dc-48e6-bb1b-14d483ec7761","resolution":{"observed_at":"2026-08-11T12:27:14.767749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.771958Z","title":"Near-optimal regret bounds for reinforcement learning","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.771958Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:bd6c8a772fd27f3740b3aa7e74423f7a036ba8204b9fc1d1e2618a0aaf676769","observation_id":"5c27494a-714e-4e5c-9bf3-44bb2e3c7c28","resolution":{"observed_at":"2026-08-11T12:27:14.771958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.776263Z","title":"Is q-learning provably efficient? Advances in neural information processing systems, 31, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.776263Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:836ea07bec272f4e2504b4c38b801623d85ea6e65b6f5f1db96ba176a593b72a","observation_id":"26401b76-537f-4e1e-918b-704f6422f873","resolution":{"observed_at":"2026-08-11T12:27:14.776263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.780488Z","title":"Is pessimism provably efficient for offline rl? In International Conference on Machine Learning, pp","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.780488Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:6e4c9ef94ce09dee95866517d64d1d77be9536a93fa20964948e4a4f1e2133b3","observation_id":"77b41dbe-7544-4752-ba91-7b270144b7ad","resolution":{"observed_at":"2026-08-11T12:27:14.780488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-11T12:27:14.784837Z","title":"Scaling laws for neural language models","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.784837Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:1915c06087f74af59e63279d896e3b8ee637b3d6aec86650531aa1ffb99ba1ef","observation_id":"3ce58390-5d87-474e-be42-7def0c668028","resolution":{"observed_at":"2026-08-11T12:27:14.784837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12309","last_updated":"2024-06-26T02:44:18Z","snapshot_observed_at":"2026-07-06T17:46:42.199436Z","submitted_at":"2024-03-18T23:18:27Z","title":"Reinforcement Learning from Delayed Observations via World Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12309","snapshot_observed_at":"2026-08-11T12:27:14.789255Z","title":"Reinforcement learning from delayed observations via world models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.789255Z"},"links":{"cited_paper":"/paper/2403.12309","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:bd91368470e61ce2568a1f9755d8e67361028d210a1217cd1e898b5a6f16f168","observation_id":"dbe37a15-1d70-4757-a357-4888eb7480a7","resolution":{"observed_at":"2026-08-11T12:27:14.789255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.793283Z","title":"Dynamic decision frequency with continuous options","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.793283Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:506ea8e2340819e0e5e3a57784f531f6db7038ed1f8b49f05762cb0ce93da2e7","observation_id":"83d15995-a8ea-4f7e-862d-cb6cada52e74","resolution":{"observed_at":"2026-08-11T12:27:14.793283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.797034Z","title":"Markov decision processes with delays and asynchronous cost collection","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.797034Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:3bbd2507134d7095d4a5ba655b59c00b76faec6116374ec59cbb6472a8989555","observation_id":"b108ebf0-0d82-4fef-84d9-30830ab0c2f4","resolution":{"observed_at":"2026-08-11T12:27:14.797034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.800965Z","title":"Domain scoping for subject matter experts","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.800965Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:96038d0a25dd295df7e41c6746264287a92dc6153c6a7366ac97b7cac587d9d1","observation_id":"087bd5a7-b645-4b38-9430-cf987ff8456f","resolution":{"observed_at":"2026-08-11T12:27:14.800965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.13490","last_updated":"2022-11-11T23:12:23Z","snapshot_observed_at":"2026-08-06T18:15:59.906402Z","submitted_at":"2020-12-25T02:35:27Z","title":"Towards Continual Reinforcement Learning: A Review and Perspectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.13490","snapshot_observed_at":"2026-08-11T12:27:14.804991Z","title":"Towards continual reinforcement learning: A review and perspectives","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.804991Z"},"links":{"cited_paper":"/paper/2012.13490","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:4551ea029140f034e4f0cd1481cb89b4e432bf2a8b9a79359713b425d8599b80","observation_id":"75a007d3-514e-428a-be07-d87fc97bcc6f","resolution":{"observed_at":"2026-08-11T12:27:14.804991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1903.03216","last_updated":"2020-05-18T15:50:48Z","snapshot_observed_at":"2026-07-06T07:37:51.655095Z","submitted_at":"2019-03-07T23:12:30Z","title":"Learning Hierarchical Teaching Policies for Cooperative Agents","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1903.03216","snapshot_observed_at":"2026-08-11T12:27:14.809102Z","title":"Learning hierarchical teaching policies for cooperative agents","venue":null,"work_id":null,"year":1903},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.809102Z"},"links":{"cited_paper":"/paper/1903.03216","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:b31863e84d1696c4609bfd334c1fd5375fbba9b94a9d093ac430d8954fbc15e5","observation_id":"f89c5bb1-9424-4a4e-9165-ed4f06ed33b1","resolution":{"observed_at":"2026-08-11T12:27:14.809102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.813510Z","title":"Hetero- geneous knowledge transfer via hierarchical teaching in cooperative multiagent reinforcement learning","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.813510Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:3a3258acebc579b5e24165e090a7ee8fe1bd63b4f4c7f8cc27abbf88468864f5","observation_id":"7e908fff-b058-45d9-8091-f9e192ae7dd0","resolution":{"observed_at":"2026-08-11T12:27:14.813510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.817747Z","title":"A policy gradient algorithm for learning to learn in multiagent reinforcement learning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.817747Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:1a9505f9f7e291fc2296e8e054230eebd4fd56291c6c6b656b38993915c9e28a","observation_id":"fb8f1726-d4f6-4b65-be0a-8faa5e205f36","resolution":{"observed_at":"2026-08-11T12:27:14.817747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.821739Z","title":"Influencing long-term behavior in multiagent reinforcement learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.821739Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:18c0bfe0642c22f792d7043ea03dba5e7f17ec636ba5d5a7dcd717edab7a13bc","observation_id":"c91f14d2-e8b4-45e4-beee-7d5e60eefea8","resolution":{"observed_at":"2026-08-11T12:27:14.821739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.16175","last_updated":"2022-10-28T14:45:39Z","snapshot_observed_at":"2026-08-03T13:44:35.048502Z","submitted_at":"2022-10-28T14:45:39Z","title":"Game-Theoretical Perspectives on Active Equilibria: A Preferred Solution Concept over Nash Equilibria","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.16175","snapshot_observed_at":"2026-08-11T12:27:14.825834Z","title":"Game-theoretical perspectives on active equilibria: A preferred solution concept over nash equilibria","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.825834Z"},"links":{"cited_paper":"/paper/2210.16175","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:dc27d7f622238306cdc05030f5fb9bf4a3ac02c618baa27341a665cbb23962cb","observation_id":"f8edd0f0-d037-4b0a-ba5f-578cc6f92c37","resolution":{"observed_at":"2026-08-11T12:27:14.825834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.830395Z","title":"Overcoming catastrophic forgetting in neural networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.830395Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:29ed61aa25c40144a1f7c811bb3505c50911bc2a787065c18904448173369db2","observation_id":"f3f95e65-dbf7-4896-9c0b-d4c1037673de","resolution":{"observed_at":"2026-08-11T12:27:14.830395Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.09437","last_updated":"2020-07-08T15:50:41Z","snapshot_observed_at":"2026-08-08T08:22:42.474573Z","submitted_at":"2020-06-16T18:29:58Z","title":"A Study of Compositional Generalization in Neural Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.09437","snapshot_observed_at":"2026-08-11T12:27:14.834450Z","title":"A study of compositional generalization in neural models","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.834450Z"},"links":{"cited_paper":"/paper/2006.09437","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:4f35402ea2a324327c6b1678b07235297fe6afa66f66853f1a3cb196774783f1","observation_id":"b8724588-9917-4033-8c29-c6bb131b118b","resolution":{"observed_at":"2026-08-11T12:27:14.834450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.08484","last_updated":"2022-06-24T20:24:35Z","snapshot_observed_at":"2026-08-10T14:40:34.682519Z","submitted_at":"2022-01-20T22:54:32Z","title":"Iterated Reasoning with Mutual Information in Cooperative and Byzantine Decentralized Teaming","version":4},"cited_work":{"arxiv_id":"2201.08484","doi":null,"metadata_source":"pith","pith_arxiv_id":"2201.08484","snapshot_observed_at":"2026-08-11T12:27:15.368246Z","title":"Iterated Reasoning with Mutual Information in Cooperative and Byzantine Decentralized Teaming","venue":"cs.MA","work_id":"2c97136b-e147-4898-a916-7dcdc7992393","year":2022},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.838998Z"},"links":{"cited_paper":"/paper/2201.08484","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:1a7bc430e668d367d7a716349aaec127972492b1137488a1d40e357eb0de8760","observation_id":"7294d8a5-f384-4ed1-9829-6b7ede3f8c5e","resolution":{"observed_at":"2026-08-11T12:27:15.374748Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.843219Z","title":"Asynchronous coagent networks","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.843219Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:fcba9284de164267d8c934723370c3b9bc067252c69ffedb56ac0959f94320cc","observation_id":"9a29402c-a35f-4eff-bb1a-96ad8a4442db","resolution":{"observed_at":"2026-08-11T12:27:14.843219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.847013Z","title":"Conservative q-learning for offline reinforcement learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.847013Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:05ff00717a618c3c9a28a36f5bc1e03cd59e21efb18ef847b1386465ac2738d0","observation_id":"eccb639e-cd4c-40f5-8cdb-49714ff3c24c","resolution":{"observed_at":"2026-08-11T12:27:14.847013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.276824Z","title":"Slow learners are fast","venue":null,"work_id":"5570878f-ba4d-475a-8c60-d3c8a2d468ff","year":2009},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.851260Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:634e048213ddae7662b31b6da6c6c26832da3d6217e5f774c56cdf185f3569b1","observation_id":"83da1e7c-a131-4ac2-bb3c-432c4a5d0382","resolution":{"observed_at":"2026-08-11T12:27:16.280996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.02419","last_updated":"2020-07-14T09:21:53Z","snapshot_observed_at":"2026-08-06T23:14:01.358514Z","submitted_at":"2020-06-03T17:50:16Z","title":"Emergent Multi-Agent Communication in the Deep Learning Era","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.02419","snapshot_observed_at":"2026-08-11T12:27:14.855190Z","title":"Emergent multi-agent communication in the deep learning era","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.855190Z"},"links":{"cited_paper":"/paper/2006.02419","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:d8e968ef96b812f77927a1505d8eb721e1151e07b3b5a168a9ccc5b916a91c75","observation_id":"e0c1c3ec-0f47-4c65-8a31-424bfcf47cc8","resolution":{"observed_at":"2026-08-11T12:27:14.855190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.263103Z","title":"Towards a unified theory of state abstraction for mdps","venue":null,"work_id":"8b3a9ead-01d9-4be8-8005-8b86d5d06e2e","year":null},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.859334Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:104c30ca2009f8b7a2d17c68b8cd83919d243b69d4f5a25e189b5c25bff904f4","observation_id":"fe7b2f65-c7c4-49fa-82f0-ccd26a83eed5","resolution":{"observed_at":"2026-08-11T12:27:16.267152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.863327Z","title":"Learning without forgetting","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.863327Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:7f37d834f217d55094377e73e5eb8b446843c0f490da671fa71a71de8573e6ea","observation_id":"9cfd5235-f7a9-4961-8853-88dbcd0b95be","resolution":{"observed_at":"2026-08-11T12:27:14.863327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.867203Z","title":"Multi-agent actor-critic for mixed cooperative-competitive environments","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.867203Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:55dc4fe512d0ddf7e87be7ce897c4cca351522843e16454267fca12e6f4da116","observation_id":"1582e272-c6d7-4a8f-abc4-230e35e0d5be","resolution":{"observed_at":"2026-08-11T12:27:14.867203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.11517","last_updated":"2020-11-23T16:28:27Z","snapshot_observed_at":"2026-08-07T12:30:11.187380Z","submitted_at":"2020-11-23T16:28:27Z","title":"Consolidation via Policy Information Regularization in Deep RL for Multi-Agent Games","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2011.11517","snapshot_observed_at":"2026-08-11T12:27:14.871555Z","title":"Consolidation via policy information regularization in deep rl for multi-agent games","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.871555Z"},"links":{"cited_paper":"/paper/2011.11517","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:c30313506a10aa2ee58a5f2a2385ea62f010ca969f37a4ec456b6f6759c1d994","observation_id":"83d5a837-bfa6-42b4-afb4-64e1c7438229","resolution":{"observed_at":"2026-08-11T12:27:14.871555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.04646","last_updated":"2020-10-09T15:42:21Z","snapshot_observed_at":"2026-07-06T10:03:00.402782Z","submitted_at":"2020-10-09T15:42:21Z","title":"Deep RL With Information Constrained Policies: Generalization in Continuous Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.04646","snapshot_observed_at":"2026-08-11T12:27:14.876277Z","title":"Deep rl with information constrained policies: Generalization in continuous control","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.876277Z"},"links":{"cited_paper":"/paper/2010.04646","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:35a31e25f91fa4084b52ed6e03c635427c3a942488b47e1005d9a641b7a2705f","observation_id":"94d8b9ad-6844-4c6f-9212-bff1a9880467","resolution":{"observed_at":"2026-08-11T12:27:14.876277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.229426Z","title":"Rl generalization in a theory of mind game through a sleep metaphor (student abstract)","venue":null,"work_id":"a5b5abd2-f586-4454-aa30-13a3a9e982fe","year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.880741Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:25a19264c13c0d9dfdcf5e701b06ac94bb330f5f0ee0f03e00971fe394634fca","observation_id":"be58ace0-fd77-46cc-b5f4-d596d4c7c5f5","resolution":{"observed_at":"2026-08-11T12:27:16.234368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.213368Z","title":"Capacity-limited decentralized actor-critic for multi-agent games","venue":null,"work_id":"b30bf607-469f-414c-83e2-293b06e4cb3d","year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.885308Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:7fd045be9568dcaca4a1afb44464fb00f7921839467f404edc1f4dc59688aa41","observation_id":"dd9c3ceb-76a4-4441-8896-0a1dd9f94a43","resolution":{"observed_at":"2026-08-11T12:27:16.218789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17508","last_updated":"2023-03-30T16:22:10Z","snapshot_observed_at":"2026-07-06T15:10:12.213443Z","submitted_at":"2023-03-30T16:22:10Z","title":"Learning in Factored Domains with Information-Constrained Visual Representations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17508","snapshot_observed_at":"2026-08-11T12:27:14.889666Z","title":"Learning in factored domains with information-constrained visual representations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.889666Z"},"links":{"cited_paper":"/paper/2303.17508","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:f9b19b449593f976a5e1c12033ac2a314101cbc70b9fee712d2b84d561be4e59","observation_id":"ba88cb19-9bf8-4ee1-bb5c-4eb1ccd6af90","resolution":{"observed_at":"2026-08-11T12:27:14.889666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.198926Z","title":"Summarizing societies: Agent abstraction in multi-agent reinforcement learning","venue":null,"work_id":"412aa33f-9e4a-4477-bb51-dde17478922b","year":2022},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.894379Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:29da4dfb4e25533f5b10e945e10db54a7d1204d394dd071cff9fbc4150d41568","observation_id":"a854065a-249f-4f60-9250-b92a158ff0a5","resolution":{"observed_at":"2026-08-11T12:27:16.203318Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:14.898919Z","title":"Human-level control through deep reinforcement learning","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.898919Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e067f2072c3e062dc36736a5d88d771b5abe29ee472fdb63788420266d134080","observation_id":"6003dc43-48a3-468a-a42c-c6d92c295c64","resolution":{"observed_at":"2026-08-11T12:27:14.898919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.175764Z","title":"Asynchronous methods for deep rein- forcement learning","venue":null,"work_id":"b606daae-9699-4e3a-b5eb-9ce09b80b226","year":1928},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.903285Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:fe992680aa0f67512dbf4f3a4eec5a56f960aa88121fcc18a2d3747443145203","observation_id":"92b42fe9-2d7c-4243-a275-2cb75521561a","resolution":{"observed_at":"2026-08-11T12:27:16.179908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.162566Z","title":"Revisiting state augmentation methods for reinforcement learning with stochastic delays","venue":null,"work_id":"8864077d-8b77-4d73-8feb-326e396c3391","year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.911492Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:7771c98ef62029022114ce1cf8f7e9a252ac99d8281be66bebabec1a157ac9dc","observation_id":"2142e7b2-56ac-4c58-b245-0f0f934c1885","resolution":{"observed_at":"2026-08-11T12:27:16.166795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.03720","last_updated":"2018-04-23T20:52:30Z","snapshot_observed_at":"2026-08-03T23:35:37.500150Z","submitted_at":"2018-04-10T21:09:53Z","title":"Gotta Learn Fast: A New Benchmark for Generalization in RL","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.03720","snapshot_observed_at":"2026-08-11T12:27:14.915508Z","title":"Gotta learn fast: A new benchmark for generalization in rl","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.915508Z"},"links":{"cited_paper":"/paper/1804.03720","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:2fcc6463394927a909aa13cdd9c56dd9d7147d5f625dae8fd7204ec86b561c17","observation_id":"e40cf72f-80c2-46b5-9dc1-fa4dfa0c973f","resolution":{"observed_at":"2026-08-11T12:27:14.915508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.01005","last_updated":"2023-06-05T14:10:00Z","snapshot_observed_at":"2026-08-11T14:57:59.684530Z","submitted_at":"2021-08-02T16:07:21Z","title":"Sequoia: A Software Framework to Unify Continual Learning Research","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.01005","snapshot_observed_at":"2026-08-11T12:27:14.920225Z","title":"Sequoia: A software framework to unify continual learning research","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.920225Z"},"links":{"cited_paper":"/paper/2108.01005","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:bc47ddaf1b5bed1cfe1a1e4c76f6cc11b3dae3479bbac7ffba64a00168e50180","observation_id":"964a3708-080b-4a5f-a9a7-fc1e0e337a40","resolution":{"observed_at":"2026-08-11T12:27:14.920225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.148814Z","title":"Learning to teach in cooperative multiagent reinforcement learning","venue":null,"work_id":"ae64e7d0-3883-497f-9d1a-e4415100c4ff","year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.924602Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:951ecdc2c42ed1598a975ed13fecd8ef3b46a84b3e7568223a20f0629b5d6fd7","observation_id":"365efc24-4f5c-449d-a27c-1caf5bfdf147","resolution":{"observed_at":"2026-08-11T12:27:16.153483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.135097Z","title":"Regret bounds for reinforcement learning via markov chain concentration","venue":null,"work_id":"77ebde5e-6eb6-4475-9824-91e3de8ec905","year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.928720Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:1b31621ae45afb466ab6e4cd693208736ca17c7783a2b75880949ffb14a4f030","observation_id":"e3e178e2-fdde-4391-b71d-ec5b3c533ade","resolution":{"observed_at":"2026-08-11T12:27:16.139550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.121506Z","title":"Comvas: Contextual moral values alignment system","venue":null,"work_id":"9ca4fd46-0433-400f-a838-042064aa8a24","year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.932722Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e88366d4e915d349cac9b61f21e0ad32bf9f8ccc24a08501d227d93db70e0d60","observation_id":"bc55efdd-e18d-40fb-939d-a23595e4aaa8","resolution":{"observed_at":"2026-08-11T12:27:16.125714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.107869Z","title":"Pytorch: An imperative style, high- performance deep learning library","venue":null,"work_id":"b97f8b8d-89a4-44b7-bb37-6f94d92819a7","year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.936808Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:be8a3c530b5ebada7899c396b2270d700731b9f37df394621620c8344213d81f","observation_id":"59619b19-2fa8-47ec-a564-554197c1acd5","resolution":{"observed_at":"2026-08-11T12:27:16.112292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.094203Z","title":"Markov decision processes","venue":null,"work_id":"343966a7-beef-4464-b41e-d0fb66038b15","year":1994},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.940874Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:3cf88d8eb16b24ac06baa28d2024dbe6c696249964f0e24fe51d5a30aff8528b","observation_id":"485f0c2d-b6fc-4abc-8224-c807355768e4","resolution":{"observed_at":"2026-08-11T12:27:16.098346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.081501Z","title":"Machine theory of mind","venue":null,"work_id":"13648f83-398d-421b-a8e3-e0abe414c238","year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.945027Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:25d7544a99bf9a7e009170e815beb51c54ea24f0e484785b430aa7eba1a60319","observation_id":"ef4834a4-75a3-47de-baaa-43a96a653fe1","resolution":{"observed_at":"2026-08-11T12:27:16.085394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.067418Z","title":"Real-time reinforcement learning","venue":null,"work_id":"7493607e-3dd9-4842-ae7c-fb940c6f6df7","year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.948899Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:8156d615ff28243c837cbd6b828f4301d4dcc65cdcfaf1e1cbb40092b9a62b80","observation_id":"00ffd810-928e-42fc-9160-d50b7f54943a","resolution":{"observed_at":"2026-08-11T12:27:16.071913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.053547Z","title":"Distributed computing in social media analytics","venue":null,"work_id":"4f554f69-1dca-4140-a434-718d4d256389","year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.952925Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:24ad3f68a69166427421996374f28552b2203709ca7c2b71dec4c869289a042a","observation_id":"1e513550-13ed-4c87-b0c2-2f7096b92e2f","resolution":{"observed_at":"2026-08-11T12:27:16.058272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.039090Z","title":"A deep learning and knowledge transfer based architecture for social media user characteristic determination","venue":null,"work_id":"3e294f51-118d-4faa-94c0-135805ee78ed","year":2015},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.956822Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e25d67fe256e355df4a96d5f32ec0e97ee041a350e81620dc6c32bd6e4350286","observation_id":"3f30e355-d817-4ef3-84d7-3090c89a9d9c","resolution":{"observed_at":"2026-08-11T12:27:16.044082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.024948Z","title":"Correcting forecasts with multifactor neural attention","venue":null,"work_id":"76181ec2-2f71-4d5c-b574-dbb7f3daabd2","year":2016},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.960748Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e43c1e02e4afd6f07e78e76584bd049850cb8554e7039d76b36e922c5d409da8","observation_id":"5b830176-b8e6-4551-ad81-f577447ab746","resolution":{"observed_at":"2026-08-11T12:27:16.029248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:16.010832Z","title":"Generative knowledge distillation for general purpose function compression","venue":null,"work_id":"60ac1357-92a8-4687-8377-66efc35d3e92","year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.964907Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:72f23d5d71592f4c4960016ed58deee0e0f5730820922320301d5bed6d3f6a84","observation_id":"61a8096b-df49-4e8e-b248-36f0d219153e","resolution":{"observed_at":"2026-08-11T12:27:16.015456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.03617","last_updated":"2017-04-12T04:38:18Z","snapshot_observed_at":"2026-07-06T05:37:31.544051Z","submitted_at":"2017-04-12T04:38:18Z","title":"Representation Stability as a Regularizer for Improved Text Analytics Transfer Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1704.03617","snapshot_observed_at":"2026-08-11T12:27:14.968725Z","title":"Representation stability as a regular- izer for improved text analytics transfer learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.968725Z"},"links":{"cited_paper":"/paper/1704.03617","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:5f283cdfb6297b3eda441c8db4aa1201dbf14ac316264ac3924296e3c11b71a7","observation_id":"a8f8927e-63b9-4476-9df8-0f23708c9c8e","resolution":{"observed_at":"2026-08-11T12:27:14.968725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.11910","last_updated":"2019-05-03T03:32:40Z","snapshot_observed_at":"2026-08-08T01:27:10.295065Z","submitted_at":"2018-10-29T00:13:50Z","title":"Learning to Learn without Forgetting by Maximizing Transfer and Minimizing Interference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.11910","snapshot_observed_at":"2026-08-11T12:27:14.973031Z","title":"Learning to learn without forgetting by maximizing transfer and minimizing interference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.973031Z"},"links":{"cited_paper":"/paper/1810.11910","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:21bab0b3eaa2f7ee1b6a37b3e67da5db7df9db46c7333bae4b830628de1b5113","observation_id":"e1a0638b-e413-4d87-8941-44a12c08e55b","resolution":{"observed_at":"2026-08-11T12:27:14.973031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.997178Z","title":"Learning abstract options","venue":null,"work_id":"51102c55-a4d9-4788-a1ae-1f9815733036","year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.977621Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:9698b19902d5e7f932c340de3706cbdae9fb797eb95cdf60b02abfdcc3a4753a","observation_id":"95d8d66a-5b04-40f3-8c63-3f20f3521bda","resolution":{"observed_at":"2026-08-11T12:27:16.001919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.983575Z","title":"Scalable recollections for continual lifelong learning","venue":null,"work_id":"084498ff-52bf-41f2-967c-fa9cd7e22aec","year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.981462Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e8195cfd8b9ce7cbeae960aebf4c3241cd8e869a3977e66f31c541a9f2505078","observation_id":"4c844ca2-6768-4694-a8a8-e9fc22383800","resolution":{"observed_at":"2026-08-11T12:27:15.987843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.969325Z","title":"On the role of weight sharing during deep option learning","venue":null,"work_id":"62bf68f9-acdb-4ecd-ba52-d3b146a5541a","year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.985584Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:7a2ebd67f49c222c8c4f79a08be8d3c838cc3f11721e17a6ddad51831337e27a","observation_id":"3d37eb38-8477-48fb-a9ba-f9312bedacf3","resolution":{"observed_at":"2026-08-11T12:27:15.973742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.949825Z","title":"Continual learning in environments with polynomial mixing times","venue":null,"work_id":"6571aefa-dd8f-402a-9d82-e8601a0297a4","year":2022},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.989450Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e5f1f0141b767a66e89f07fff280d32e39456ff40e9f2be3414ec36b87a1c1e2","observation_id":"0738d394-5aa3-4e8e-9362-4cfc7fa0009c","resolution":{"observed_at":"2026-08-11T12:27:15.954100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.936367Z","title":"Balancing context length and mixing times for reinforcement learning at scale","venue":null,"work_id":"72d68909-0829-42f4-86b7-97b4236af7a3","year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.993355Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:5d273c3e0fe25c2226fe9ca9cca97e08d239dc8187e45dd28686bbb67208b08a","observation_id":"35cb7252-1ffa-4c2b-85ff-a4fa89b20e2c","resolution":{"observed_at":"2026-08-11T12:27:15.940664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.922640Z","title":"Realtime reinforcement learning: Towards rapid asynchronous deployment of large models","venue":null,"work_id":"f859dbf0-839d-4cb6-bd04-d89c3f9ee2c7","year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:14.997330Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:81c83535da432977d21b4d2e96a310477597d4857fe2f4d1762a65ad802292b2","observation_id":"45d37930-939e-46b0-8809-2d22b0d76147","resolution":{"observed_at":"2026-08-11T12:27:15.926945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.907954Z","title":"Routing networks: Adaptive selection of non-linear functions for multi-task learning","venue":null,"work_id":"aa231706-bb0f-4ffe-a893-92437cefc5e1","year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.001208Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:0b5f7c20230cd35c4418129d59128c326afae02cc84fbbca436ee2bf4cb24253","observation_id":"14ae1d6f-3604-4a1e-8a74-3eff9f09b904","resolution":{"observed_at":"2026-08-11T12:27:15.912708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.892613Z","title":"Dispatched routing networks","venue":null,"work_id":"5321b0c7-90cf-4b3f-b004-20454b3174e0","year":2019},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.005734Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e8773d20335b62a301c1700f6bd05baab564c6a8b44c21af5b04b06dee08dbd7","observation_id":"8c2b864e-8597-455f-a980-70684bf74bfb","resolution":{"observed_at":"2026-08-11T12:27:15.897118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.12774","last_updated":"2019-04-29T15:32:14Z","snapshot_observed_at":"2026-07-06T07:49:15.217918Z","submitted_at":"2019-04-29T15:32:14Z","title":"Routing Networks and the Challenges of Modular and Compositional Computation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.12774","snapshot_observed_at":"2026-08-11T12:27:15.010184Z","title":"Routing networks and the challenges of modular and compositional computation","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.010184Z"},"links":{"cited_paper":"/paper/1904.12774","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:54fe45b2c594a002dff421732ee50d315e1abb676024340422a02af4c47f5f50","observation_id":"f9594823-3c5d-4c40-9b8e-d448449301d9","resolution":{"observed_at":"2026-08-11T12:27:15.010184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.877913Z","title":"Control delay in reinforcement learning for real-time dynamic systems: A memoryless approach","venue":null,"work_id":"9ff4e198-caf7-4ae9-9e61-992664a03b07","year":2010},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.014901Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:dc5339067b3dcb3576f349b887f2fbc1b0ad12971370fd3b466a12d8e5a9db82","observation_id":"8f4188cf-bed3-4482-a278-a647c36e1b64","resolution":{"observed_at":"2026-08-11T12:27:15.882923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-11T12:27:15.018952Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.018952Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:4fc7f6c24e1aba93f2cc381746806e0c46a791e95e8cedcdb43d0e6aa2ceb9a0","observation_id":"3f84a9fa-23e7-4b1a-bf67-d66d1b12ef4f","resolution":{"observed_at":"2026-08-11T12:27:15.018952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.864327Z","title":"Bigger, better, faster: Human-level atari with human-level efficiency","venue":null,"work_id":"6e8536eb-52fd-4855-986f-d637e7312e20","year":2023},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.023116Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:d048953e2a56c0e05ac32f1fe399202d9b3a3e8a23404677a7dd8da438a5d283","observation_id":"55e8fdd2-de7d-425c-b16e-e8b46d8855e4","resolution":{"observed_at":"2026-08-11T12:27:15.868449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.851068Z","title":"Pokémon red/blue/rng manipulation faq, 2020","venue":null,"work_id":"87d48acf-4487-4029-9c02-9386cfebdd2e","year":2020},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.027245Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:00847c4ef83649b41f9a9d15e179af31872b8290ea1f42892f22ab9542e0ae5d","observation_id":"b7ca989e-4e7b-4154-9904-c9d6755c90e7","resolution":{"observed_at":"2026-08-11T12:27:15.855244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.837061Z","title":"Pokemon red/blue leaderboard, 2024","venue":null,"work_id":"3ff61eaa-486e-4126-b885-bb0deea1bc58","year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.031036Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:1ff6918f6caffa763755f2604ea5c3fb61fc6704a92eef8bbac6ab846eaca89d","observation_id":"5604b39a-e6e6-4edd-ae16-9714f802bc66","resolution":{"observed_at":"2026-08-11T12:27:15.841314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.034810Z","title":"Reinforcement learning: An introduction","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.034810Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:93e7e22f116f0e9926cb8f74ef6ee23a362835e149e695ccbf63a9b9d3a86e62","observation_id":"34eefcfc-d87e-45bb-b237-007c867ccbbd","resolution":{"observed_at":"2026-08-11T12:27:15.034810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.038803Z","title":"Between mdps and semi-mdps: A framework for temporal abstraction in reinforcement learning","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.038803Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:8af6deff1a243f16127ee5401aaa05bb43aebdd77527b0d10672c4c38373cdc7","observation_id":"75eb3a9b-b0f5-4a05-996a-bf558d451f03","resolution":{"observed_at":"2026-08-11T12:27:15.038803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.803835Z","title":"Horde: A scalable real-time architecture for learning knowledge from unsupervised sensorimotor interaction","venue":null,"work_id":"5c5c08f3-ffb3-499a-970f-a8e9d2856ca5","year":2011},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.042627Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:63a831aa6a555422bd684ab6c204c9d9d3ead843203d2862f5177353023dd5bd","observation_id":"5e576380-2621-4e92-8b46-1f5e3252e11b","resolution":{"observed_at":"2026-08-11T12:27:15.808522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04879","last_updated":"2024-06-07T12:25:51Z","snapshot_observed_at":"2026-08-07T10:12:21.002117Z","submitted_at":"2024-06-07T12:25:51Z","title":"A Deep Dive into the Trade-Offs of Parameter-Efficient Preference Alignment Techniques","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04879","snapshot_observed_at":"2026-08-11T12:27:15.046470Z","title":"A deep dive into the trade-offs of parameter-efficient preference alignment techniques","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.046470Z"},"links":{"cited_paper":"/paper/2406.04879","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:9a1bcede3170cb1d497c25b7d16e954d777c96ffd80a379fd868a8b56cd41255","observation_id":"5647d77d-8fcb-4be3-9ed2-b50b3c8d46ed","resolution":{"observed_at":"2026-08-11T12:27:15.046470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.06824","last_updated":"2025-05-30T17:17:04Z","snapshot_observed_at":"2026-08-05T09:29:42.302441Z","submitted_at":"2024-11-11T09:32:20Z","title":"Combining Domain and Alignment Vectors to Achieve Better Knowledge-Safety Trade-offs in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.06824","snapshot_observed_at":"2026-08-11T12:27:15.050862Z","title":"Combining domain and alignment vectors to achieve better knowledge-safety trade-offs in llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.050862Z"},"links":{"cited_paper":"/paper/2411.06824","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:09808e13fad4031305426fc387bb0c69a0e3aff88ac19cd031dc1dba7e61a525","observation_id":"0bc06504-d1ea-4892-811f-4bd2bfe5d1cf","resolution":{"observed_at":"2026-08-11T12:27:15.050862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.790389Z","title":null,"venue":null,"work_id":"3db1370e-0fc1-4384-9472-d2bc59291941","year":2011},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.054937Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:d1597025d6be24677574472f6dce65e032fedebeaf023245e6cfeb0ddfb9ddc4","observation_id":"d5b35cfe-96bd-4be9-b3c7-92771cc55c1a","resolution":{"observed_at":"2026-08-11T12:27:15.794586Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T12:27:15.776218Z","title":"Scalable approaches for a theory of many minds","venue":null,"work_id":"93300a5c-513b-4715-9590-65053da62afa","year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.058713Z"},"links":{"citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:e87912f073bf117e0daad502b6f770026ee88c0f032b8193eaf2b8d5cd0ddeca","observation_id":"3db06a8e-4974-486a-be49-7b6f79c7b686","resolution":{"observed_at":"2026-08-11T12:27:15.781218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.17032","last_updated":"2025-11-02T13:42:19Z","snapshot_observed_at":"2026-08-12T13:10:57.260306Z","submitted_at":"2024-07-24T06:35:05Z","title":"Gymnasium: A Standard Interface for Reinforcement Learning Environments","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.17032","snapshot_observed_at":"2026-08-11T12:27:15.062636Z","title":"Gymnasium: A standard interface for reinforcement learning environments","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-11T12:27:15.062636Z"},"links":{"cited_paper":"/paper/2407.17032","citing_paper":"/paper/2412.14355"},"observation_digest":"sha256:739cff1851a7564ad5381cb3cbb9270db94e504e0e738d65e180f7f672284272","observation_id":"80348ff6-5ca7-4d79-86b4-96438e8153d7","resolution":{"observed_at":"2026-08-11T12:27:15.062636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.14355","last_updated":"2024-12-18T21:43:40Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-11T19:04:40.952532Z","submitted_at":"2024-12-18T21:43:40Z","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":67,"verified_exact":1,"verified_fuzzy":32},"total_outbound_references":112},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 100 of 112 outbound references and 4 inbound Pith citation observations for arXiv:2412.14355."}