{"as_of":"2026-08-07T17:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:67b8c2f2839929f14f4b4a98e3874426fc451c2e2403a45ee65ccc5b381b9c65","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T23:21:13.658826Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T00:18:12.785448Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-03T17:58:47.598667Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"cited_work":{"arxiv_id":"2603.29844","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.29844","snapshot_observed_at":"2026-07-03T17:58:47.598667Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","venue":"cs.RO","work_id":"e0e30762-359c-4722-9272-c41480fc67fd","year":2026},"citing_paper":{"arxiv_id":"2605.14712","last_updated":"2026-07-11T13:09:54Z","snapshot_observed_at":"2026-07-16T23:17:51.777870Z","submitted_at":"2026-05-14T11:31:02Z","title":"IntentVLA: Short-Horizon Intent Modeling for Aliased Robot Manipulation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-30T20:55:56.318157Z"},"links":{"cited_paper":"/paper/2603.29844","citing_paper":"/paper/2605.14712"},"observation_digest":"sha256:5f5542671f8334aed53feb8e6531d1e6e7e573c1fa0c8c9f9b6079224a5e7ee0","observation_id":"05b73946-9726-4fb7-a1a7-53f13247b360","resolution":{"observed_at":"2026-07-01T14:35:46.767295Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.29844","snapshot_observed_at":"2026-07-14T18:59:49.358909Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.14712","last_updated":"2026-07-11T13:09:54Z","snapshot_observed_at":"2026-07-16T23:17:51.777870Z","submitted_at":"2026-05-14T11:31:02Z","title":"IntentVLA: Short-Horizon Intent Modeling for Aliased Robot Manipulation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-14T18:59:49.358909Z"},"links":{"cited_paper":"/paper/2603.29844","citing_paper":"/paper/2605.14712"},"observation_digest":"sha256:adede4744d58b407771b1f9fe34b91742222e7ac7088269d1d8e4816b36c5033","observation_id":"c4f45136-24d0-49db-b4b2-a63f41f1dd0b","resolution":{"observed_at":"2026-07-14T18:59:49.358909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"cited_work":{"arxiv_id":"2603.29844","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.29844","snapshot_observed_at":"2026-07-03T17:58:47.598667Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","venue":"cs.RO","work_id":"e0e30762-359c-4722-9272-c41480fc67fd","year":2026},"citing_paper":{"arxiv_id":"2606.00113","last_updated":"2026-05-27T05:32:17Z","snapshot_observed_at":"2026-08-01T09:58:43.654749Z","submitted_at":"2026-05-27T05:32:17Z","title":"World Models for Robotic Manipulation: A Survey","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-29T12:24:18.025364Z"},"links":{"cited_paper":"/paper/2603.29844","citing_paper":"/paper/2606.00113"},"observation_digest":"sha256:26b7ca3b8ac6704bd3f81482f33d58e13d2efa1b9874bd96fa0258d27124f49f","observation_id":"569809f2-468a-4b66-bc89-99f587b32288","resolution":{"observed_at":"2026-06-29T12:33:25.001319Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"cited_work":{"arxiv_id":"2603.29844","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.29844","snapshot_observed_at":"2026-07-03T17:58:47.598667Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","venue":"cs.RO","work_id":"e0e30762-359c-4722-9272-c41480fc67fd","year":2026},"citing_paper":{"arxiv_id":"2606.12217","last_updated":"2026-06-10T15:31:25Z","snapshot_observed_at":"2026-08-02T10:56:16.670279Z","submitted_at":"2026-06-10T15:31:25Z","title":"Making Foresight Actionable: Repurposing Representation Alignment in World Action Models","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-27T09:40:54.963759Z"},"links":{"cited_paper":"/paper/2603.29844","citing_paper":"/paper/2606.12217"},"observation_digest":"sha256:ad44c182f18ae0dbbe809ee32c7018b80825c89feace75f3ba8d152c67b27686","observation_id":"3674c4fb-e992-4a3b-88d4-22353dcda17e","resolution":{"observed_at":"2026-07-03T11:08:03.376362Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"cited_work":{"arxiv_id":"2603.29844","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.29844","snapshot_observed_at":"2026-07-03T17:58:47.598667Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","venue":"cs.RO","work_id":"e0e30762-359c-4722-9272-c41480fc67fd","year":2026},"citing_paper":{"arxiv_id":"2606.17200","last_updated":"2026-06-15T18:40:18Z","snapshot_observed_at":"2026-08-04T22:44:21.289174Z","submitted_at":"2026-06-15T18:40:18Z","title":"ACE-Ego-0: Unifying Egocentric Human and Robotic Data for VLA Pretraining","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T03:25:39.450667Z"},"links":{"cited_paper":"/paper/2603.29844","citing_paper":"/paper/2606.17200"},"observation_digest":"sha256:fc1e49263277a2d051263402f3357e261969536167d98be4c0faba77b4e971db","observation_id":"f7914f00-e4d1-4c9e-8a25-2fdb018b5f55","resolution":{"observed_at":"2026-07-03T17:58:47.599986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.29844","snapshot_observed_at":"2026-07-14T05:52:34.589171Z","title":"Dial: Decoupling intent and action via latent world modeling for end-to-end vla,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.11397","last_updated":"2026-07-13T11:02:04Z","snapshot_observed_at":"2026-08-03T16:19:12.630473Z","submitted_at":"2026-07-13T11:02:04Z","title":"WALA Learning Executable Latent Actions from Action-Labeled Demonstrations and Action-Free Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-14T05:52:34.589171Z"},"links":{"cited_paper":"/paper/2603.29844","citing_paper":"/paper/2607.11397"},"observation_digest":"sha256:a9a0c48a68bdc0322561995afbea1b5517782160a642738bae86162adef54efe","observation_id":"f7bb466b-d7c2-499c-8d49-404f97cb8a5f","resolution":{"observed_at":"2026-07-14T05:52:34.589171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.29844","snapshot_observed_at":"2026-08-06T00:18:12.785448Z","title":"Dial: Decoupling intent and action via latent world modeling for end-to-end vla, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.01397","last_updated":"2026-08-02T17:25:08Z","snapshot_observed_at":"2026-08-06T23:26:21.196822Z","submitted_at":"2026-08-02T17:25:08Z","title":"SG-WAM: Self-Guided World Modeling in Geometry-Aware Policy Space","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T00:18:12.785448Z"},"links":{"cited_paper":"/paper/2603.29844","citing_paper":"/paper/2608.01397"},"observation_digest":"sha256:edf43b8b27264a31393b459a0671bd1c8672ba0e13523eef86c1623bbce54e69","observation_id":"27bdb77f-5671-436e-ab04-abf083406e04","resolution":{"observed_at":"2026-08-06T00:18:12.785448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2603.29844/citation-record","integrity":"/paper/2603.29844/integrity","json":"/paper/2603.29844/citation-record.json","paper":"/paper/2603.29844"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Paligemma: A versatile 3b vlm for transfer","venue":null,"work_id":"9d7438fe-8d33-4d9f-8f22-c97d40149bca","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:e2a58ec0a433a2dd7ddce3358e605f613bb8f5f31846e4a09ff2acfc1f1a2742","observation_id":"3c39a5be-70c7-44ee-bcf6-eaef48518a25","resolution":{"observed_at":"2026-05-13T23:23:27.233536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Eagle 2.5: Boosting long-context post-training for frontier vision-language models","venue":null,"work_id":"dded38e0-417e-4f4c-9584-561d68bce7a1","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:883353568e82388cbbb57453d46d9835873fc3a0599d1183a27ae0fb6df494c3","observation_id":"477cf41a-6477-4e92-8357-7dabb987e61b","resolution":{"observed_at":"2026-05-13T23:23:27.239454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:dc7e1e93e77969016f1f5e23b509a324bf067373a1bc0020d23f4b025f5765fe","observation_id":"c97a815f-26f2-41a9-b817-7d10daff85dd","resolution":{"observed_at":"2026-05-13T23:23:26.359919Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:fcff51db4ddcd854d4853cc813d6e667e5314374714714b3eb3e8681c6398aac","observation_id":"10662f23-2d2e-4719-a889-c0717e91495d","resolution":{"observed_at":"2026-05-13T23:23:26.373508Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":"2307.15818","doi":"10.48550/arxiv.2307.15818","metadata_source":"pith","pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","venue":"cs.RO","work_id":"ff438a8a-8003-4fae-9131-acd418b3597b","year":2023},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:fc4ac466a35c1ed8e1e730961688f26607d48d927da52b3ab83b775a88bf6dea","observation_id":"41ee3560-7973-4805-a6b6-f52f964f8130","resolution":{"observed_at":"2026-05-13T23:23:26.316201Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"184dad8a-114d-469d-b953-e6a96d5ab968","year":2023},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:d8350d4683046a276882333d11bfedd772e1a9a04ca3f53ae99c53123b1cc723","observation_id":"130f669f-1b53-4008-beee-e2cccb4c89d9","resolution":{"observed_at":"2026-05-13T23:23:27.288239Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openvla: An open-source vision-language-action model","venue":null,"work_id":"8efb129e-8419-4d11-ae0a-21be14ff3ff0","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:f2d6e5d855ad6fda94f93a87553aab17e4c2c4b91c819428e9836a35ee1ac347","observation_id":"320752ee-44ae-4abb-8b11-f91afe4f7793","resolution":{"observed_at":"2026-05-13T23:23:27.284060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.01691","last_updated":"2022-08-16T16:06:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-04T17:57:11Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","version":2},"cited_work":{"arxiv_id":"2204.01691","doi":"10.48550/arxiv.2204.01691","metadata_source":"pith","pith_arxiv_id":"2204.01691","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","venue":"cs.RO","work_id":"037320f1-b0a9-4cbe-a639-bfb25409ce71","year":2022},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2204.01691","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:d2dab542eba78ea7d8e7ae91ac9b8e408084b1c9033b0f2679959e1e0de19723","observation_id":"08df3935-4f4c-4522-8cc2-8107d7641b6d","resolution":{"observed_at":"2026-05-13T23:23:26.338436Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hi robot: Open-ended instruction following with hierarchical vision-language-action models","venue":null,"work_id":"56fbc89a-cfb0-4b9b-97e5-89f83032a005","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:a149d33b6e95f52ffd3be776766e9065455463eab024a4d67e275a97c68285d6","observation_id":"aa030518-5b9b-43c2-a442-e72197ecf0e8","resolution":{"observed_at":"2026-05-13T23:23:27.344894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rekep: Spatio- temporal reasoning of relational keypoint constraints for robotic manipulation","venue":null,"work_id":"f4f1617a-7f82-4254-8d8e-d7d99ad82940","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:28fb62c7c39ea2ee8472bf95a9181ea9793149388c7efa2b74fa75c03312589b","observation_id":"bcc66cfe-005a-403f-922f-d303a32b7a03","resolution":{"observed_at":"2026-05-13T23:23:27.275571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"GR00T N1: An open foundation model for generalist humanoid robots","venue":null,"work_id":"8aec8d81-ffc7-446d-91af-efc275ea8d01","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:08828409666b9eaf0c41e3e4a11e434559b4775f121b0b0a640bfb12de4a8786","observation_id":"6459839d-a1d2-4aa6-91e5-b25a7af9973f","resolution":{"observed_at":"2026-05-13T23:23:27.258802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8f461a2b-a734-40af-9c7b-ff37c6962dda","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:d936e221c79d383e82d9ae72c070d4af5301147e464ee91bdb05abdbab0fdeb7","observation_id":"82abb517-cc54-4d53-a026-cb52d080aa6c","resolution":{"observed_at":"2026-05-13T23:23:27.356722Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15659","last_updated":"2025-05-21T15:33:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T15:33:27Z","title":"FLARE: Robot Learning with Implicit World Modeling","version":1},"cited_work":{"arxiv_id":"2505.15659","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15659","snapshot_observed_at":"2026-07-10T12:47:05.520966Z","title":"FLARE: Robot Learning with Implicit World Modeling","venue":"cs.RO","work_id":"0734908a-7122-4eeb-ba5d-944092bb8897","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2505.15659","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:a62e36cc216c599a4325308678b51c3f53934e052f6d11f24a1b0c225faaeb35","observation_id":"1975b9a0-d6a5-4297-bf65-91f775fd62ad","resolution":{"observed_at":"2026-05-17T15:59:09.111136Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cot-vla: Visual chain-of-thought reasoning for vision-language- action models","venue":null,"work_id":"7aa059ce-2c61-4954-a912-2103a6fbff3d","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:eeb166a8e7d81603ed35f6fc626fe3aeab01ddad411f27023c6d262d73619f2e","observation_id":"ef6a66b3-6b61-4990-9a82-7f86f09ded74","resolution":{"observed_at":"2026-05-13T23:23:27.292108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15109","last_updated":"2024-12-19T17:52:50Z","snapshot_observed_at":"2026-07-06T20:10:19.704368Z","submitted_at":"2024-12-19T17:52:50Z","title":"Predictive Inverse Dynamics Models are Scalable Learners for Robotic Manipulation","version":1},"cited_work":{"arxiv_id":"2412.15109","doi":"10.48550/arxiv.2412.15109","metadata_source":"pith","pith_arxiv_id":"2412.15109","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Predictive Inverse Dynamics Models are Scalable Learners for Robotic Manipulation","venue":"cs.RO","work_id":"50911eef-b866-4e23-b501-1b4c4bfd1a78","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2412.15109","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:a2feceee148a53ef5c781426b3650801ec679cdaa0eb52ebe0ef1ca0cace10e5","observation_id":"ef532a97-aa2a-48c0-8b6d-bb3e6de5c121","resolution":{"observed_at":"2026-05-22T14:38:25.165602Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":"880c8731-d932-4c0f-9dd5-3cfdf30bfa41","year":2023},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:409964a2a14be8c3037518790d2649e4e798246c96ac9460e2253ba28de626d0","observation_id":"2dbaf191-8cfb-491f-b7ff-47d83fdb3c2a","resolution":{"observed_at":"2026-05-13T23:23:27.333055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:54:50.449643Z","title":"Visual instruction tuning","venue":null,"work_id":"0a395ab5-dfdd-4489-ae3c-594c4431acd9","year":2023},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:7a840f9a88dcbcb9f237f25dce343135e22c3d3ad68b4c18dd806f29cc68fe82","observation_id":"feca6a88-0b21-4716-9a40-5d2e81364b2b","resolution":{"observed_at":"2026-05-13T23:23:27.262142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Egoplan-bench: Benchmarking multimodal large language models for human-level planning","venue":null,"work_id":"6a0999e8-d62b-43e6-8625-eb678b876c99","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:ed4fc28c397185e45b5e6a058f3bc74a35bfe1c9fbea1c1816ba783e695fee28","observation_id":"ab1e0bf0-efac-4dbb-af5c-4a66de53cb65","resolution":{"observed_at":"2026-05-13T23:23:27.267171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Code as policies: Language model programs for embodied control","venue":null,"work_id":"4cb0f69e-ebac-46ec-a00d-201cc9cfa3a0","year":2023},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:c53c37ee27e235fac3cd6368a609483a0b073fe4358036cc44230b2e8417f762","observation_id":"09802b67-2016-49bf-8eb9-7a55f4edd7a9","resolution":{"observed_at":"2026-05-13T23:23:27.322027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b01314f9-5be9-4437-92e0-8e1038cdede9","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:dc57b53000a5e21e2a3adf58bbb349fea7a4db78437415223a22590d6afc4d3a","observation_id":"c0a6ab37-e5c3-49ce-8082-37ccd02adaad","resolution":{"observed_at":"2026-05-13T23:23:27.318283Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tenenbaum, Dale Schuurmans, and Pieter Abbeel","venue":null,"work_id":"226445b9-f1db-4da8-a6ca-af93af40de9e","year":2023},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:93eb6dc653ef00f25de23cc56e28e85ab9824470ad6b2f967e9c679d36655d02","observation_id":"6b64f0c0-890a-4ebf-bfb4-0f0c843a94bc","resolution":{"observed_at":"2026-05-13T23:23:27.314392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09747","last_updated":"2025-01-16T18:57:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T18:57:04Z","title":"FAST: Efficient Action Tokenization for Vision-Language-Action Models","version":1},"cited_work":{"arxiv_id":"2501.09747","doi":"10.48550/arxiv.2501.09747","metadata_source":"pith","pith_arxiv_id":"2501.09747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"FAST: Efficient Action Tokenization for Vision-Language-Action Models","venue":"cs.RO","work_id":"83a8f966-6cfa-4f21-81f3-87440aae238f","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2501.09747","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:0d550bf8c241b8b055c71e88fda759766672f7b32e11e6aa21e40325956b7a0c","observation_id":"e024f619-ddaa-41a3-a3c6-f57b528a210e","resolution":{"observed_at":"2026-05-13T23:23:26.310458Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19645","last_updated":"2025-04-28T07:49:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-27T00:30:29Z","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","version":2},"cited_work":{"arxiv_id":"2502.19645","doi":"10.48550/arxiv.2502.19645","metadata_source":"pith","pith_arxiv_id":"2502.19645","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","venue":"cs.RO","work_id":"04f46bb3-4346-47e8-bf09-c75d91f96e87","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2502.19645","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:17344aa8075f37035ff580cf1a3b22f0da3a4d3c3ac62364f910b6d676aadec1","observation_id":"a8907712-3fa3-46f5-8067-1cc56465e088","resolution":{"observed_at":"2026-05-13T23:23:26.366771Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":"2410.24164","doi":"10.48550/arxiv.2410.24164","metadata_source":"pith","pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","venue":"cs.LG","work_id":"f790abdc-a796-482f-a40d-f8ee035ecfc2","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:2693206e4a4f91ce014829f2268401fa8368dde354e8f4765ea97371e8d5aa7f","observation_id":"6b057b30-2481-42f0-a3fe-187af6413d85","resolution":{"observed_at":"2026-05-13T23:23:26.353798Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ren, Homer Walke, Quan Vuong, Lucy Xiaoyang Shi, and Sergey Levine","venue":null,"work_id":"91655bed-7b48-43a5-9d26-e3360c1416b7","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:91723a938b14110ab0c1f7979a2a6b83f959cfd6cc5dbd48a859e5fa8bddb4de","observation_id":"ba92cf27-12f7-454c-b563-6e50ba677222","resolution":{"observed_at":"2026-05-13T23:23:27.242994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gr00t n1.6: An im- proved open foundation model for generalist humanoid robots","venue":null,"work_id":"a9491be5-0db4-4aa6-8986-3988502dbfd1","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:854dd979638292dbe1b6d02c076a5ddcfd86a32f2d6e8e0af30f919ee043062c","observation_id":"73017f00-d469-4d35-a890-b7d40381b75a","resolution":{"observed_at":"2026-05-13T23:23:27.300114Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.19650","last_updated":"2024-11-29T12:06:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-29T12:06:03Z","title":"CogACT: A Foundational Vision-Language-Action Model for Synergizing Cognition and Action in Robotic Manipulation","version":1},"cited_work":{"arxiv_id":"2411.19650","doi":"10.48550/arxiv.2411.19650","metadata_source":"pith","pith_arxiv_id":"2411.19650","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CogACT: A Foundational Vision-Language-Action Model for Synergizing Cognition and Action in Robotic Manipulation","venue":"cs.RO","work_id":"4b158d3e-3dff-4412-85cd-baa879465a5e","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2411.19650","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:de3c07cd6a6772c6a2e694102745b05e6785e92c00e008b7cbd45eddb3981dd3","observation_id":"27da5bd5-abe3-4225-984f-1a0e694a3056","resolution":{"observed_at":"2026-05-13T23:23:26.297923Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gr-3 technical report","venue":null,"work_id":"59d92f51-ab52-4510-ab07-71af81937847","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:431714ea298bb493089e413a3a53fe68226d5e6ea9032883d96fd71af6f35d1d","observation_id":"72301bd2-c9e4-4fbc-a4e9-0a7d41641a42","resolution":{"observed_at":"2026-05-13T23:23:27.250618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Igniting vlms toward the embodied space","venue":null,"work_id":"4c31df96-b928-4970-b8e8-8ad150ed16d4","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:647cc0607f0cf0ec1d2acdcbbaf5df305b49b4b67632230852a0146bf1dd7327","observation_id":"625e70ec-5d19-4d0f-8cc4-9281cd0f4629","resolution":{"observed_at":"2026-05-13T23:23:27.271890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Robotic control via embodied chain-of-thought reasoning","venue":null,"work_id":"ab3f23d6-ea75-4348-a0c7-66adb9531c0d","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:a8f199533000ecb0f6f3b039305ccb9dcf05aed22590b7dcc15c5c6ab57b2a33","observation_id":"af34a71e-f9c8-4be9-9936-de7a55153b92","resolution":{"observed_at":"2026-05-13T23:23:27.279812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Molmoact: Action reasoning models that can reason in space","venue":null,"work_id":"c331ac18-681e-49e5-b79d-68fb90521916","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:941844c7def24a11ae86205932551a47ae2420d8e12d88040d289eb7702648c9","observation_id":"326a1e34-73bd-44ad-a249-5eb2a894d3d2","resolution":{"observed_at":"2026-05-13T23:23:27.296012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unleashing large-scale video generative pre-training for visual robot manipulation","venue":null,"work_id":"eb67ddaa-166c-4c1f-bd63-26c30a19cc34","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:6f9ccead983d0b932a1d7b5338c1927cc6d32d95e8d6ba169b998ffce04567cd","observation_id":"e0ef737a-7dde-40a3-b87f-e8e0c6f86add","resolution":{"observed_at":"2026-05-13T23:23:27.329101Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gr-2: A generative video-language-action model with web-scale knowledge for robot manipulation","venue":null,"work_id":"f2deabd0-1411-4555-aabd-0555e0dfc84e","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:1c908f0e3bc3520b033f38c8473a6df650d2d116bc3f9bafdaf1a7e6d38bd7fa","observation_id":"492a9166-0654-48c0-a2a4-fcb43b5c76eb","resolution":{"observed_at":"2026-05-13T23:23:27.340623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unified vision-language-action model","venue":null,"work_id":"1fdf3431-ce13-48f1-95a7-1e5c2062c6fe","year":2026},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:7fa5eeb23d1859d087af49060601d5240e43b7e0ed0fac9fc35dfeb4a13f5f5c","observation_id":"2cd4a1cc-27dc-4930-9df8-a3148d86e8bb","resolution":{"observed_at":"2026-05-13T23:23:27.352874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Worldvla: Towards autoregressive action world model","venue":null,"work_id":"e1830e5b-bced-4ae6-9936-67b808fcf609","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:da0ec68ca8511633dfc37b4374310a3dda7c7fd019649d516d571060695c738e","observation_id":"8af9dca0-a714-4bb6-b95f-4081082cb8c1","resolution":{"observed_at":"2026-05-13T23:23:27.349029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Latent action pretraining from videos","venue":null,"work_id":"1696e37d-8a84-406b-a56c-72ffdbc7dcd4","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:915dc699f5834aaf312d70ac58709f48963817b3b988eb016be33c8166ccaa88","observation_id":"d664ce26-54b0-4117-a20c-906aa1370dea","resolution":{"observed_at":"2026-05-13T23:23:27.325486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Moto: Latent motion token as the bridging language for learning robot manipulation from videos","venue":null,"work_id":"6d1f3a4b-4b30-4ad5-b296-c2b29ef9de14","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:2576a73f9b8cf671cf0af0b206e8c791beb207a438797d78eaa254d3d1ab29b9","observation_id":"d32d7c11-c71e-4a70-86a5-ec165b3956ed","resolution":{"observed_at":"2026-05-13T23:23:27.337010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"villa- x: Enhancing latent action modeling in vision-language-action models","venue":null,"work_id":"cc56b621-2b0c-460c-a24a-e48f3adaaab3","year":2026},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:3f16ad58fd7538b9ba221b235952c9b11a26e5873ea1ebe9a542d148efe5bb12","observation_id":"1bf737ad-d08e-43fa-a1b0-71572a753e9c","resolution":{"observed_at":"2026-05-13T23:23:27.254966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unicod: Enhancing robot policy via unified continuous and discrete representation learning","venue":null,"work_id":"9fac8bff-f4fc-4a5d-b794-581731c4b8e1","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:fd3aef6b432b86a63832a6d1cf327daa395fc63bf31050a59ca1b45110a114fe","observation_id":"8d38d9e6-9be3-4e04-bdde-a4d777f4ef97","resolution":{"observed_at":"2026-05-13T23:23:27.246738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11709","last_updated":"2026-03-09T06:44:42Z","snapshot_observed_at":"2026-07-06T21:25:23.371749Z","submitted_at":"2025-05-16T21:34:47Z","title":"EgoDex: Learning Dexterous Manipulation from Large-Scale Egocentric Video","version":3},"cited_work":{"arxiv_id":"2505.11709","doi":"10.48550/arxiv.2505.11709","metadata_source":"pith","pith_arxiv_id":"2505.11709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"EgoDex: Learning Dexterous Manipulation from Large-Scale Egocentric Video","venue":"cs.CV","work_id":"4190fb9c-70e0-4388-aa0b-516589489047","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2505.11709","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:24753add9d03f65dab2fd06ed93357cfd1fa46b662ad88591969f23e129eddd4","observation_id":"aafb4aae-2b89-4ff4-8d87-188f2f7d6a72","resolution":{"observed_at":"2026-05-15T15:40:29.471447Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Diffusion policy: Visuomotor policy learning via action diffusion","venue":null,"work_id":"8d3517d4-ba69-4deb-8b5f-43c1a97b4ae7","year":2023},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:b6a1b133a9fdb1ec56bd176c4f905c22f4367992dda211e6a68574ecf2794b86","observation_id":"3a3941dc-c1a2-46ae-903a-895fdf060210","resolution":{"observed_at":"2026-05-13T23:23:27.310904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.00200","last_updated":"2025-04-24T20:02:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-28T21:38:17Z","title":"Unified Video Action Model","version":3},"cited_work":{"arxiv_id":"2503.00200","doi":"10.48550/arxiv.2503.00200","metadata_source":"pith","pith_arxiv_id":"2503.00200","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Unified Video Action Model","venue":"cs.RO","work_id":"fb4cc512-d1d9-40f4-8854-d35950ad20b3","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"cited_paper":"/paper/2503.00200","citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:89ecfa02f01e0aab366a17e2bee38393fe06d41d709bef858ff297e0adfcbf15","observation_id":"66bd4548-52a5-4c9f-bfcc-b7db01df7cb7","resolution":{"observed_at":"2026-05-13T23:23:26.321209Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Starvla: A lego-like codebase for vision-language-action model develop- ing","venue":null,"work_id":"aace2f53-216d-4aa6-a7f6-d36a0f1e51a2","year":2025},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:a2795c7a9617073f23b19fe5cf9fa8113ed362cd7991bdb5abfe57593a0721d8","observation_id":"c8d7c8f8-64b2-4fa1-af31-fd53ca519555","resolution":{"observed_at":"2026-05-13T23:23:27.303706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":"0234c1f5-6599-4d3a-a402-0268740d8865","year":2024},"citing_paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T23:21:13.658826Z"},"links":{"citing_paper":"/paper/2603.29844"},"observation_digest":"sha256:7b477e1faf672141d49481e9bb94288563d8749a858d8da6d83995a5028cc562","observation_id":"164e0c9b-221d-4c91-a9f2-8ce7d5fb794f","resolution":{"observed_at":"2026-05-13T23:23:27.307206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2603.29844","last_updated":"2026-04-28T02:10:56Z","latest_version":2,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-01T00:31:11.395521Z","submitted_at":"2026-03-31T15:02:27Z","title":"DIAL: Decoupling Intent and Action via Latent World Modeling for End-to-End VLA"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":3,"verified_exact":12,"verified_fuzzy":29},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 7 inbound Pith citation observations for arXiv:2603.29844."}