{"as_of":"2026-08-08T21:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b799b76c0e0a75b8d19565d0e670dfb7c3622039acd916cd2cddaad799b48e9c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":31,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:23:17.808710Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T03:56:35.270996Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2503.03480","last_updated":"2026-04-19T06:23:17Z","snapshot_observed_at":"2026-07-30T14:07:10.544745Z","submitted_at":"2025-03-05T13:16:55Z","title":"SafeVLA: Towards Safety Alignment of Vision-Language-Action Model via Constrained Learning","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-23T01:27:33.123243Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2503.03480"},"observation_digest":"sha256:5d34cfab68091ff4d3cc96ee80d8339d99f27f16e58cca5976bbc519724055c4","observation_id":"ce718d6f-8e18-4186-b3aa-289210808907","resolution":{"observed_at":"2026-05-23T01:32:22.648354Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-07T14:23:17.808710Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model.arXiv preprint arXiv:2502.14420,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19080","last_updated":"2025-05-25T10:24:44Z","snapshot_observed_at":"2026-08-08T15:03:07.282678Z","submitted_at":"2025-05-25T10:24:44Z","title":"ReFineVLA: Reasoning-Aware Teacher-Guided Transfer Fine-Tuning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:23:17.808710Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2505.19080"},"observation_digest":"sha256:edc4eb62c5d73807c7df3d2e5a2fa2cc67a73c09a90749f7dff5c8c19dba0c56","observation_id":"9dc6f745-b6fe-4d6f-ac3e-d374ec8ffdc1","resolution":{"observed_at":"2026-08-07T14:23:17.808710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-07T13:25:57.438562Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model.arXiv preprint arXiv:2502.14420, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.21906","last_updated":"2025-05-29T23:34:24Z","snapshot_observed_at":"2026-08-07T13:17:24.769477Z","submitted_at":"2025-05-28T02:48:42Z","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T13:25:57.438562Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2505.21906"},"observation_digest":"sha256:689e628afc8343dad0f5c9bfc625024c85c4b8c66507238171c4f3881b533f8c","observation_id":"226371c4-8467-4e3d-a7c1-017f7d6f1c8c","resolution":{"observed_at":"2026-08-07T13:25:57.438562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-07T12:49:35.637019Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23450","last_updated":"2025-06-11T03:16:06Z","snapshot_observed_at":"2026-08-07T12:43:12.816737Z","submitted_at":"2025-05-29T13:56:49Z","title":"Agentic Robot: A Brain-Inspired Framework for Vision-Language-Action Models in Embodied Agents","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T12:49:35.637019Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2505.23450"},"observation_digest":"sha256:f30c226dce6a968a2f883ff01e0c4e5eaa7ee12de8bc9fd8adbdea2e6ce8aa53","observation_id":"15ee6829-a0a9-4ab3-a46f-971d4b4f73f7","resolution":{"observed_at":"2026-08-07T12:49:35.637019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-07T12:16:16.632623Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language- action model","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00123","last_updated":"2025-05-30T18:00:34Z","snapshot_observed_at":"2026-08-08T15:03:06.735553Z","submitted_at":"2025-05-30T18:00:34Z","title":"Visual Embodied Brain: Let Multimodal Large Language Models See, Think, and Control in Spaces","version":1},"reference_index":110,"source":"pdf_text","source_observed_at":"2026-08-07T12:16:16.632623Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2506.00123"},"observation_digest":"sha256:712828ab64a41cc0bddf34f0e10ec61cd3e3ddbf52c4a30d802795af6d2db453","observation_id":"6082cc5c-280e-499e-852e-b1f654eeb5af","resolution":{"observed_at":"2026-08-07T12:16:16.632623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-07T00:40:38.147800Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.13045","last_updated":"2025-08-24T10:01:04Z","snapshot_observed_at":"2026-08-08T06:33:39.031465Z","submitted_at":"2025-06-16T02:27:25Z","title":"Continual Learning for Generative AI: From LLMs to MLLMs and Beyond","version":4},"reference_index":271,"source":"pdf_text","source_observed_at":"2026-08-07T00:40:38.147800Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2506.13045"},"observation_digest":"sha256:733f74b01104bad10a88550ae0e991c82bf98f37924599ddbf64441945d2e6af","observation_id":"31675d45-1f75-4d09-ad2e-e1456f0170f9","resolution":{"observed_at":"2026-08-07T00:40:38.147800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-07T00:31:31.745215Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model.arXiv preprint arXiv:2502.14420, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.13679","last_updated":"2025-06-16T16:34:20Z","snapshot_observed_at":"2026-08-07T13:45:40.850146Z","submitted_at":"2025-06-16T16:34:20Z","title":"ROSA: Harnessing Robot States for Vision-Language and Action Alignment","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:31:31.745215Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2506.13679"},"observation_digest":"sha256:1755c6f0d6e17991d3fd9b23b5fda9f4367b64c6cef8aa5c7b529fde02b5013b","observation_id":"165a2863-afd9-46ef-9a11-eeb195aa3bcd","resolution":{"observed_at":"2026-08-07T00:31:31.745215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2507.04447","last_updated":"2025-08-26T08:23:50Z","snapshot_observed_at":"2026-08-05T16:56:52.400110Z","submitted_at":"2025-07-06T16:14:29Z","title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","version":3},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-05-16T15:42:41.363422Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2507.04447"},"observation_digest":"sha256:55added44c5bbddfff401d9d937e2f74668c536128c5e8831c587fa2465e5b0f","observation_id":"7750dc80-a3b8-4e5e-b028-6508e0dda973","resolution":{"observed_at":"2026-05-16T15:42:41.600535Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2508.13073","last_updated":"2025-09-01T08:10:01Z","snapshot_observed_at":"2026-08-07T15:40:31.068428Z","submitted_at":"2025-08-18T16:45:48Z","title":"Large VLM-based Vision-Language-Action Models for Robotic Manipulation: A Survey","version":2},"reference_index":130,"source":"pdf_text","source_observed_at":"2026-05-17T20:28:15.818016Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2508.13073"},"observation_digest":"sha256:ab0f7f3716741cdda45f2a3d34ba95a10111736c22f4a2b3f665f7e1372b59ce","observation_id":"7e44a777-c2af-4e07-a7c5-f5b07db981bc","resolution":{"observed_at":"2026-05-17T20:28:16.340673Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-05T16:55:52.091468Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17449","last_updated":"2025-09-04T16:46:34Z","snapshot_observed_at":"2026-08-05T16:55:50.271610Z","submitted_at":"2025-08-24T17:01:15Z","title":"Robotic Manipulation via Imitation Learning: Taxonomy, Evolution, Benchmark, and Challenges","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T16:55:52.091468Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2508.17449"},"observation_digest":"sha256:7be0619d6f29d9db1de27173fd5249175a886cc6621613ce770a3326148e0c56","observation_id":"9a1b0fe7-1d5d-4f37-8fab-cf43743bbc1a","resolution":{"observed_at":"2026-08-05T16:55:52.091468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-05T12:02:28.260522Z","title":"Learning to act anywhere with task-centric latent actions.arXiv preprint arXiv:2502.14420,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.02055","last_updated":"2025-09-05T06:24:50Z","snapshot_observed_at":"2026-08-06T20:42:39.880464Z","submitted_at":"2025-09-02T07:51:59Z","title":"Align-Then-stEer: Adapting the Vision-Language Action Models through Unified Latent Guidance","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-05T12:02:28.260522Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2509.02055"},"observation_digest":"sha256:390074edeffdec4e0a24b0b5d59de10a2a73ad22238cd35aec619557b28ca6e4","observation_id":"5590b66d-51b2-45ab-8e61-386d196ec944","resolution":{"observed_at":"2026-08-05T12:02:28.260522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-04T22:55:30.078793Z","title":"Chatvla: Unified multimodal un- derstanding and robot control with vision-language-action model.arXiv preprint arXiv:2502.14420, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.06932","last_updated":"2025-09-10T14:34:25Z","snapshot_observed_at":"2026-08-07T11:30:34.307197Z","submitted_at":"2025-09-08T17:45:40Z","title":"LLaDA-VLA: Vision Language Diffusion Action Models","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-04T22:55:30.078793Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2509.06932"},"observation_digest":"sha256:34304492f4e2693694c567d05c6166dd5ebc2e343f142bc7e63149c9093757fc","observation_id":"ee066cac-abb4-4331-9d3d-4ca7f09bc4e9","resolution":{"observed_at":"2026-08-04T22:55:30.078793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-04T12:54:50.518362Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.01642","last_updated":"2026-07-07T04:25:23Z","snapshot_observed_at":"2026-08-08T01:28:25.872832Z","submitted_at":"2025-10-02T03:48:07Z","title":"FailSafe: Reasoning and Recovery from Failures in Vision-Language-Action Models","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T12:54:50.518362Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2510.01642"},"observation_digest":"sha256:35782b8aaf2d6cbede2d98e8b4a08e72009f457e54872e2e03e735be4ffc227e","observation_id":"95a960aa-0b45-45b7-816e-f31416027001","resolution":{"observed_at":"2026-08-04T12:54:50.518362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-04T12:55:08.357317Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model.arXiv preprint arXiv:2502.14420,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.01711","last_updated":"2026-05-31T14:16:48Z","snapshot_observed_at":"2026-08-04T12:55:01.902418Z","submitted_at":"2025-10-02T06:41:22Z","title":"Contrastive Representation Regularization for Vision-Language-Action Models","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T12:55:08.357317Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2510.01711"},"observation_digest":"sha256:e16df2d19367be63392bda5bb25861831dfca35a37024b9053bc490018aaf49d","observation_id":"81a5e571-eef7-4a7d-8bfe-cc1b12e59f34","resolution":{"observed_at":"2026-08-04T12:55:08.357317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-03T20:59:53.480873Z","title":"ChatVLA: Unified multimodal understanding and robot control with vision-language-action model.arXiv preprint arXiv:2502.14420,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.17502","last_updated":"2026-05-30T03:49:39Z","snapshot_observed_at":"2026-08-03T20:59:48.866420Z","submitted_at":"2025-11-21T18:59:32Z","title":"RynnVLA-002: A Unified Vision-Language-Action and World Model","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T20:59:53.480873Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2511.17502"},"observation_digest":"sha256:537c52382afc06089409459fc984f73ac1bdb0c240c0f686510239cb41a853de","observation_id":"d35bf74b-9c79-41c5-a2b3-3b570a5605dc","resolution":{"observed_at":"2026-08-03T20:59:53.480873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-03T13:53:24.338053Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.22539","last_updated":"2026-07-02T08:55:28Z","snapshot_observed_at":"2026-08-06T03:03:38.906157Z","submitted_at":"2025-12-27T09:40:54Z","title":"VLA-Arena: An Open-Source Framework for Benchmarking Vision-Language-Action Models","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T13:53:24.338053Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2512.22539"},"observation_digest":"sha256:cabf105001f0202a526ed4dd9f4a8eb129a0f8823f5ef6fe640ae4add4f865b6","observation_id":"714b5cbe-822d-4707-80fa-1c8fb1481890","resolution":{"observed_at":"2026-08-03T13:53:24.338053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-03T13:29:48.142773Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.24125","last_updated":"2026-07-26T09:02:49Z","snapshot_observed_at":"2026-08-08T10:23:37.779496Z","submitted_at":"2025-12-30T10:18:42Z","title":"Unified Embodied VLM Reasoning with Robotic Action via Autoregressive Discretized Pre-training","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T13:29:48.142773Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2512.24125"},"observation_digest":"sha256:b50ca5db1a6ec3c7cf00de437ca66b82d9125e31361cdd9aa6cc67de26f902f2","observation_id":"9970abb1-2bbe-41cf-9bd2-0c321d4a3f31","resolution":{"observed_at":"2026-08-03T13:29:48.142773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2601.21998","last_updated":"2026-03-22T15:37:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-29T17:07:43Z","title":"Causal World Modeling for Robot Control","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T13:53:52.188890Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2601.21998"},"observation_digest":"sha256:33039be548ee8cde5ed06e898245a373809ee0c442a31b90a7896d0d17f948cd","observation_id":"52e7d8cf-2367-4189-8a01-9161a1456139","resolution":{"observed_at":"2026-05-12T13:53:52.273673Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-03T01:21:31.938466Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.11885","last_updated":"2026-07-31T03:33:21Z","snapshot_observed_at":"2026-08-08T15:03:07.920934Z","submitted_at":"2026-02-12T12:34:56Z","title":"Choose What to Manipulate: Revealing Data Scaling Laws in Bounding-Box Guided Policies for Semantic Manipulation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T01:21:31.938466Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2602.11885"},"observation_digest":"sha256:f77270ddf57d61fea6c48fc2d153aaefb04f0ca5b63ded5cf1cce4f401705257","observation_id":"4603c4c1-bc1c-4a54-ae48-c84a8ea3e876","resolution":{"observed_at":"2026-08-03T01:21:31.938466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2602.20231","last_updated":"2026-04-09T04:26:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-23T18:41:41Z","title":"UniLACT: Depth-Aware RGB Latent Action Learning for Vision-Language-Action Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-15T20:18:31.988002Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2602.20231"},"observation_digest":"sha256:1426ff077c925b960ed6ba02fc46b48d4d386b5ec42b96c59173d7dd39ee8119","observation_id":"25a6971c-98df-42d9-8f0a-5688ce22e5dd","resolution":{"observed_at":"2026-05-15T20:20:17.660663Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2604.15483","last_updated":"2026-04-24T23:18:28Z","snapshot_observed_at":"2026-08-05T18:18:04.172934Z","submitted_at":"2026-04-16T19:18:07Z","title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T11:42:34.409651Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2604.15483"},"observation_digest":"sha256:c1153d31cd9d34ed096a0adc9e0479035c64a38c3d51cc625036d9d93c19a37d","observation_id":"a3ede2b6-4afd-416e-8adc-8a04621e0239","resolution":{"observed_at":"2026-05-10T11:45:21.885175Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2604.17800","last_updated":"2026-04-20T04:46:20Z","snapshot_observed_at":"2026-07-06T23:04:50.935335Z","submitted_at":"2026-04-20T04:46:20Z","title":"ReFineVLA: Multimodal Reasoning-Aware Generalist Robotic Policies via Teacher-Guided Fine-Tuning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T05:06:38.517652Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2604.17800"},"observation_digest":"sha256:289e255a2d86b95dfb9c1d01e4aa2fecd3027289ff3e9e0e6fce9fc450a06728","observation_id":"c89e4ba4-0911-4812-8646-87c3cad1273c","resolution":{"observed_at":"2026-05-10T09:48:48.378335Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2605.10821","last_updated":"2026-07-16T12:45:34Z","snapshot_observed_at":"2026-08-02T14:25:58.642997Z","submitted_at":"2026-05-11T16:37:34Z","title":"UniSteer: Unified Noise Steering for Efficient Human-Guided VLA Adaptation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T04:08:43.222818Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2605.10821"},"observation_digest":"sha256:05c740a3d9d406ad1ac0dc254c5a5818f98b8ff4777d5c5c7d17a5eb694520f3","observation_id":"b41e6278-0791-4b48-a7c1-92779fb0379b","resolution":{"observed_at":"2026-05-12T06:36:25.599144Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-02T14:25:59.915385Z","title":"Learning to act anywhere with task-centric latent actions.arXiv preprint arXiv:2502.14420, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.10821","last_updated":"2026-07-16T12:45:34Z","snapshot_observed_at":"2026-08-02T14:25:58.642997Z","submitted_at":"2026-05-11T16:37:34Z","title":"UniSteer: Unified Noise Steering for Efficient Human-Guided VLA Adaptation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T14:25:59.915385Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2605.10821"},"observation_digest":"sha256:b8252cac1e70ebd88988d777eb8356ce0575b793f86059b9ce5dcb6509bc98f3","observation_id":"2601a711-a270-4ee6-a3b8-75518bd1020b","resolution":{"observed_at":"2026-08-02T14:25:59.915385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2605.11665","last_updated":"2026-07-28T11:54:22Z","snapshot_observed_at":"2026-08-02T14:19:47.159631Z","submitted_at":"2026-05-12T07:26:39Z","title":"Nautilus: From One Prompt to Plug-and-Play Robot Learning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-13T01:05:44.188530Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2605.11665"},"observation_digest":"sha256:09870daed63d07f44dfa653559b368226d2f60dbbe934d1b86e6b2de49cc66c8","observation_id":"fa7d8c17-1731-43a7-b7fe-455644114272","resolution":{"observed_at":"2026-05-13T01:07:00.091658Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-02T14:19:53.160490Z","title":"Chatvla: Unified multimodal understanding and robot control with vision-language-action model, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.11665","last_updated":"2026-07-28T11:54:22Z","snapshot_observed_at":"2026-08-02T14:19:47.159631Z","submitted_at":"2026-05-12T07:26:39Z","title":"Nautilus: From One Prompt to Plug-and-Play Robot Learning","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-02T14:19:53.160490Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2605.11665"},"observation_digest":"sha256:94b8eeb80d3de7ed6b206249ddb18e3b5fbad2231c34ac980a8ffe29820feae3","observation_id":"94d3015b-a23f-4590-b598-68fbad6a88d0","resolution":{"observed_at":"2026-08-02T14:19:53.160490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2605.20082","last_updated":"2026-05-19T16:36:34Z","snapshot_observed_at":"2026-08-03T04:22:47.012823Z","submitted_at":"2026-05-19T16:36:34Z","title":"VL-DPO: Vision-Language-Guided Finetuning for Preference-Aligned Autonomous Driving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T05:39:04.567741Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2605.20082"},"observation_digest":"sha256:0b0b9ea45b849793a4ca5c4791c43e481564c028b05e973d05abed89c1cac212","observation_id":"e17c54e4-9524-42c0-83ab-f12a4be8cdd6","resolution":{"observed_at":"2026-05-20T05:43:06.012813Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2606.00110","last_updated":"2026-05-27T03:38:15Z","snapshot_observed_at":"2026-08-06T22:19:24.528783Z","submitted_at":"2026-05-27T03:38:15Z","title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","version":1},"reference_index":138,"source":"arxiv_source","source_observed_at":"2026-06-29T13:33:03.368006Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2606.00110"},"observation_digest":"sha256:d6c4fc0f60ae1b4d8e4f0784927235540b0b4a960a5ef3578fb31f0abdc75bf9","observation_id":"2ef5d5d5-c09b-4a28-9d18-af9f70033dd6","resolution":{"observed_at":"2026-06-29T13:33:27.764526Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":"2502.14420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-07-02T03:56:35.270996Z","title":"Learning to act anywhere with task-centric latent actions","venue":null,"work_id":"ede707b9-3ea7-4c02-8162-5bfcdb7a8a1a","year":2025},"citing_paper":{"arxiv_id":"2606.03392","last_updated":"2026-06-02T09:34:08Z","snapshot_observed_at":"2026-08-08T02:32:41.958431Z","submitted_at":"2026-06-02T09:34:08Z","title":"OpenEAI-Platform: An Open-source Embodied Artificial Intelligence Hardware-Software Unified Platform","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T09:31:56.289129Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2606.03392"},"observation_digest":"sha256:30ffd4055500a8f1bbe29f0c91ff95e93b2aff8048e50e700f9e7975ef904a26","observation_id":"dc42698d-7344-40de-8ccc-e74059df9d02","resolution":{"observed_at":"2026-07-02T03:56:35.273554Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-02T05:16:43.132982Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13429","last_updated":"2026-07-15T04:13:54Z","snapshot_observed_at":"2026-08-07T12:37:37.871440Z","submitted_at":"2026-07-15T04:13:54Z","title":"Generalizable VLA Finetuning via Representation Anchoring and Language-Action Alignment","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-02T05:16:43.132982Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2607.13429"},"observation_digest":"sha256:9e137a76458d87d6ba8f40d6a52f2ff02160087d7791169fb20cac8fd6aecc18","observation_id":"5d4e160d-c8c2-441f-ad35-0690867a9bb5","resolution":{"observed_at":"2026-08-02T05:16:43.132982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14420","snapshot_observed_at":"2026-08-01T14:39:43.165360Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18709","last_updated":"2026-07-22T05:33:17Z","snapshot_observed_at":"2026-08-08T18:57:26.098275Z","submitted_at":"2026-07-21T05:05:01Z","title":"RoboInter1.5: A Holistic Intermediate Representation Suite for Embodied World Modeling and Robotic Manipulation","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-01T14:39:43.165360Z"},"links":{"cited_paper":"/paper/2502.14420","citing_paper":"/paper/2607.18709"},"observation_digest":"sha256:be1b658b27e131e06c709d12b3f2a76c24a0a68bc438e7119af49538cf7a113e","observation_id":"084a202f-25d2-4bdf-8987-b7df96f0cfc3","resolution":{"observed_at":"2026-08-01T14:39:43.165360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.14420/citation-record","integrity":"/paper/2502.14420/integrity","json":"/paper/2502.14420/citation-record.json","paper":"/paper/2502.14420"},"outbound":[],"paper":{"arxiv_id":"2502.14420","last_updated":"2025-02-21T07:28:36Z","latest_version":2,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-08T15:02:42.425911Z","submitted_at":"2025-02-20T10:16:18Z","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 31 inbound Pith citation observations for arXiv:2502.14420."}