{"as_of":"2026-08-04T19:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0134ade7cde5f79f3840c83b4737c3d57fafcd47718debbf203a02a1252bd061","coverage":[{"denominator":17,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T21:10:38.484582Z","state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T08:53:31.123181Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.17091","snapshot_observed_at":"2026-08-01T08:53:31.123181Z","title":"Learning to plan, planning to learn: Adaptive hierarchical RL-MPC for sample-eﬃcient decision making,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20973","last_updated":"2026-07-23T06:50:59Z","snapshot_observed_at":"2026-08-01T08:53:25.756220Z","submitted_at":"2026-07-23T06:50:59Z","title":"Deep Reinforcement-Learning-Guided Model Predictive Control for Preventing Overtakes in Autonomous Racing","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T08:53:31.123181Z"},"links":{"cited_paper":"/paper/2512.17091","citing_paper":"/paper/2607.20973"},"observation_digest":"sha256:a5a07815c18f14163c409b9c967a78ae00a12954963b45ae6feff477776ab16f","observation_id":"10bc8b39-f458-4823-b588-dfd021e47d3c","resolution":{"observed_at":"2026-08-01T08:53:31.123181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2512.17091/citation-record","integrity":"/paper/2512.17091/integrity","json":"/paper/2512.17091/citation-record.json","paper":"/paper/2512.17091"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":"1606.01540","doi":"10.1109/jssc.2019","metadata_source":"pith","pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"OpenAI Gym","venue":"cs.LG","work_id":"6af98f3f-f074-41ae-a689-7dd7b4b8efde","year":2016},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:17afd1c6c1e1f8e637cc404875017691402b4407d15aeeda596ce836a26dd649","observation_id":"89493903-9a99-473b-bd31-69f08f0dc9be","resolution":{"observed_at":"2026-05-16T21:11:16.778701Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1706.01502","last_updated":"2017-11-07T20:45:59Z","snapshot_observed_at":"2026-07-06T05:45:39.485902Z","submitted_at":"2017-06-05T19:01:26Z","title":"UCB Exploration via Q-Ensembles","version":3},"cited_work":{"arxiv_id":"1706.01502","doi":null,"metadata_source":"pith","pith_arxiv_id":"1706.01502","snapshot_observed_at":"2026-07-04T03:09:30.514483Z","title":"UCB Exploration via Q-Ensembles","venue":"cs.LG","work_id":"5e090ae9-8be7-4f94-a260-9093d69dd328","year":2017},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/1706.01502","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:c7f41ec8fbc5c94a605a1d992aeac499ab1095d6ea079f5dcc2ffcb9cd4cf04b","observation_id":"c2a29e33-de26-42e6-82cb-2e8a6753f5a4","resolution":{"observed_at":"2026-05-16T21:11:16.782360Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Td-mpc2: Scalable, robust world models for continuous control","venue":null,"work_id":"6f943319-896d-48c4-8600-567bf3f31e7a","year":2024},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:e19d06e4eef1a692615f55aeaea23853881281a42589a44c866680a670844ba5","observation_id":"889510dd-7412-4fe0-8376-98e4ce3c289a","resolution":{"observed_at":"2026-05-16T21:11:17.233074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16828","last_updated":"2024-03-21T17:56:19Z","snapshot_observed_at":"2026-07-31T05:32:29.431480Z","submitted_at":"2023-10-25T17:57:07Z","title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","version":2},"cited_work":{"arxiv_id":"2310.16828","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.16828","snapshot_observed_at":"2026-07-08T02:04:26.234596Z","title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","venue":"cs.LG","work_id":"360ec5fb-79fd-4490-bc73-3d161609c42d","year":2023},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/2310.16828","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:d67c9cb340d2e361ec7dd3643c3f98c03ccbcbdb9c3268f751fa80f8af361f58","observation_id":"e391eb19-83b1-46e9-95a4-ca85572e563b","resolution":{"observed_at":"2026-05-16T21:11:16.814065Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.20706","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Real-time gait adaptation for quadrupeds using model predictive control and reinforcement learning.arXiv preprint arXiv:2510.20706","venue":null,"work_id":"447b4ebf-85df-4542-8c20-679e043bbbd6","year":null},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:420c04fb2038adbcaead64a529fdab50b26bcbb3347b79eaed941e571747cac7","observation_id":"7b080c14-6b01-42eb-ae29-1bbe6dd8a330","resolution":{"observed_at":"2026-05-16T21:11:16.795150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20476","last_updated":"2025-03-04T13:11:38Z","snapshot_observed_at":"2026-07-06T20:43:59.574176Z","submitted_at":"2025-02-27T19:26:36Z","title":"Unifying Model Predictive Path Integral Control, Reinforcement Learning, and Diffusion Models for Optimal Control and Planning","version":2},"cited_work":{"arxiv_id":"2502.20476","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.20476","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unifying model predictive path integral control, reinforcement learning, and diffu- sion models for optimal control and planning.arXiv preprint arXiv:2502.20476","venue":null,"work_id":"bdb85b76-4f8b-4c87-b4a2-fdd4a0d50c6d","year":null},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/2502.20476","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:cba9ead5542009bec25ab9cc8650169004d39fe15b72cb05909e8b7e505e1fe1","observation_id":"b59c09ec-9c11-48b0-96f7-51ad64493e79","resolution":{"observed_at":"2026-05-16T21:11:16.806116Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.10470","last_updated":"2021-08-25T23:42:59Z","snapshot_observed_at":"2026-07-06T11:40:56.544714Z","submitted_at":"2021-08-24T01:38:11Z","title":"Isaac Gym: High Performance GPU-Based Physics Simulation For Robot Learning","version":2},"cited_work":{"arxiv_id":"2108.10470","doi":"10.48550/arxiv.2108.10470","metadata_source":"pith","pith_arxiv_id":"2108.10470","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Isaac Gym: High Performance GPU-Based Physics Simulation For Robot Learning","venue":"cs.RO","work_id":"a21210c8-5b8f-429a-accc-fb4ca1efd19d","year":2021},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/2108.10470","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:87f5127ddd7d0665289d5dac32874febc5060622f1295b6c64b9c8d14ad81aa5","observation_id":"9b2d108c-c455-4b83-96be-81a7c3c95c00","resolution":{"observed_at":"2026-05-16T21:11:16.810107Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1561/2200000086","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T02:02:59.242429Z","title":"Moerland, Joost Broekens, Aske Plaat, and Catholijn M","venue":null,"work_id":"1806f59f-c226-4b71-8c64-8b5bc363c8f7","year":2023},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:a5a33728b6884e6764ea7042b17c9dda806d5d46ae2e6298b5ec0a38d93ac3d6","observation_id":"e95f15e8-a5c2-4aef-8e82-d9b827f48e9a","resolution":{"observed_at":"2026-05-16T21:11:16.357775Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-13T20:20:58.806708+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T20:20:58.806708+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13549","last_updated":"2025-05-19T04:32:14Z","snapshot_observed_at":"2026-07-06T21:26:35.442603Z","submitted_at":"2025-05-19T04:32:14Z","title":"TD-GRPC: Temporal Difference Learning with Group Relative Policy Constraint for Humanoid Locomotion","version":1},"cited_work":{"arxiv_id":"2505.13549","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.13549","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Td- grpc: Temporal difference learning with group relative policy constraint for humanoid locomotion.arXiv preprint arXiv:2505.13549","venue":null,"work_id":"eb58c8c6-c4c9-4ec3-bafd-bd3a3d9f6364","year":null},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/2505.13549","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:54aaeeee2cf152b8f0cec004f2ff9f17b93277c268a631bed792f678e87ac03a","observation_id":"f65740bf-e880-4694-a923-f7fcc0acd06a","resolution":{"observed_at":"2026-05-16T21:11:16.790943Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2306.09852","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Actor-critic model predictive control: Differentiable optimization meets reinforcement learn- ing","venue":null,"work_id":"faf05741-cab5-438e-aec7-54014795dbe0","year":2024},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:5c8642d78a1ef38889eb36d773743ba6905e7309c8bd254e59c4b5061948ed01","observation_id":"d9266b73-c226-402c-a52d-e0b4edcc0c09","resolution":{"observed_at":"2026-05-16T21:11:16.819087Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1506.02438","last_updated":"2018-10-20T18:55:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2015-06-08T11:12:48Z","title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation","version":6},"cited_work":{"arxiv_id":"1506.02438","doi":"10.48550/arxiv.1506.02438","metadata_source":"pith","pith_arxiv_id":"1506.02438","snapshot_observed_at":"2026-07-11T03:57:46.887146Z","title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation","venue":"cs.LG","work_id":"38e3ca94-96f0-4b19-a355-0754931af8be","year":2015},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/1506.02438","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:b6dd1151de9a2486024e223f1cc3c5c4b2943a7e0117e6f1754b4becc38873f3","observation_id":"17906c10-c23e-47c4-b177-effce9049956","resolution":{"observed_at":"2026-05-16T21:11:16.822808Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":"1712.01815","doi":"10.48550/arxiv.1712.01815","metadata_source":"pith","pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-07-10T14:07:06.673451Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","venue":"cs.AI","work_id":"9978bd70-b9dd-4eb6-928b-66c2d40da222","year":2017},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:2faa4048a9842c827c0d2e0cb107525c52c7176ddf47a6417a6d8335d8abc35e","observation_id":"1a2b69ba-8b4f-4e69-b893-e00b80a8a1c1","resolution":{"observed_at":"2026-05-16T21:11:16.802600Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T13:38:12.382503+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2022.320734","doi":"10.1109/tnnls.2022.3207346","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Yubin Wang, Zengqi Peng, Yusen Xie, Yulin Li, Hakim Ghazzai, and Jun Ma","venue":null,"work_id":"1fd06875-118e-4907-a674-c995eaa108ec","year":2024},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:cfe70d0029aa3940a0c91d9268904ba1e271bd3cc7715117aad4ef4ec956af80","observation_id":"5e26ba27-c60b-4ac5-abba-bb997a5fcd8a","resolution":{"observed_at":"2026-05-16T21:11:16.353694Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15808","last_updated":"2024-02-28T17:31:01Z","snapshot_observed_at":"2026-08-01T19:47:35.118567Z","submitted_at":"2023-08-30T07:23:37Z","title":"Learning the References of Online Model Predictive Control for Urban Self-Driving","version":2},"cited_work":{"arxiv_id":"2308.15808","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15808","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Yuhang Wang, Hanwei Guo, Sizhe Wang, Long Qian, and Xuguang Lan","venue":null,"work_id":"69b611ee-e7b3-4b2c-b796-2ee43cfc6426","year":null},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/2308.15808","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:40dd6fd5ff239d3b76145b5b18016fc5c6f7c5d2d74600286c1de981f50e7a30","observation_id":"2a910ddc-52fe-47f8-b80d-5b345c1eaa9d","resolution":{"observed_at":"2026-05-16T21:11:16.799416Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.04280","last_updated":"2026-05-20T20:46:35Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T16:45:38Z","title":"A KL-regularization Framework for Learning to Plan with Adaptive Priors","version":2},"cited_work":{"arxiv_id":"2510.04280","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2510.04280","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A kl-regularization framework for learning to plan with adaptive priors.arXiv preprint, arXiv:2510.04280","venue":null,"work_id":"1a101f8b-e647-48a6-9924-c58618f3ba95","year":null},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"cited_paper":"/paper/2510.04280","citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:5e25f451368598f3b6bfbde8ce1d30b1ebed05c1a3bc325409cd4c3acff4d15d","observation_id":"9971c900-d870-435b-ac2e-c08eb1f4ef07","resolution":{"observed_at":"2026-05-22T02:03:27.082124Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"• Passing reward: increased with the vehicle’s relative speed to a nearbyOtherVehiclewhen overtaking (i.e., larger forward relative velocity yields larger reward)","venue":null,"work_id":"e1fe44fe-d66f-4968-973e-e0a954f0bfc8","year":2017},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:efefa16d5559fc0db26a2de6848cf7525c6f932c32399ec9af6e83e3b09e2920","observation_id":"07131240-f5e1-40ea-97c3-c2165f0b1a9c","resolution":{"observed_at":"2026-05-16T21:11:17.238714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"RL term.Three common forms are used for the RL term","venue":null,"work_id":"008c7475-1896-4501-889a-57694d0b730f","year":2048},"citing_paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T21:10:38.484582Z"},"links":{"citing_paper":"/paper/2512.17091"},"observation_digest":"sha256:26ad2871249fb650dd1a20414ec4693d3a7988fbc5f847a0a9bb55843e830b14","observation_id":"3563ce6c-d570-4e30-bf70-805a17e327da","resolution":{"observed_at":"2026-05-16T21:11:17.235724Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2512.17091","last_updated":"2026-04-16T02:13:06Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T22:39:28.855621Z","submitted_at":"2025-12-18T21:44:00Z","title":"Learning to Plan, Planning to Learn: Adaptive Hierarchical RL-MPC for Sample-Efficient Decision Making"},"reference_resolution":{"displayed":17,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":0,"verified_exact":11,"verified_fuzzy":2},"total_outbound_references":17},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 17 of 17 outbound references and 1 inbound Pith citation observation for arXiv:2512.17091."}