{"as_of":"2026-08-03T12:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b9112aa9f5da2ba8a57b9e98e551cf788c5dac40d2e2566253e8a5cb96a9612a","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T05:09:21.028373Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-03T06:30:56.289259+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T01:34:31.166730Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.10485","snapshot_observed_at":"2026-08-02T01:34:31.166730Z","title":"VEGA: Visual encoder grounding alignment for spatially-aware vision-language-action models,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14635","last_updated":"2026-07-16T06:59:39Z","snapshot_observed_at":"2026-08-02T01:34:27.769203Z","submitted_at":"2026-07-16T06:59:39Z","title":"Action QFormer: Structured Representation Shaping under Action Supervision in Vision-Language-Action Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T01:34:31.166730Z"},"links":{"cited_paper":"/paper/2605.10485","citing_paper":"/paper/2607.14635"},"observation_digest":"sha256:9821df7b518beabf596431e3e8769192a29ffcf44f393b4c8bfc7cd6fa5e6272","observation_id":"9637b991-567d-4564-8175-23d732540a7d","resolution":{"observed_at":"2026-08-02T01:34:31.166730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.10485/citation-record","integrity":"/paper/2605.10485/integrity","json":"/paper/2605.10485/citation-record.json","paper":"/paper/2605.10485"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.05800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:08:58.820608Z","title":"3d cavla: Leveraging depth and 3d context to generalize vision language action models for unseen tasks","venue":null,"work_id":"10a5d041-6d20-404d-a379-a0782b16f45e","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:196345f0478c602cb8dfaa789a6805942fd1930a40bac3a5d8fb6cfd3ce360a0","observation_id":"288c0cb6-07f0-4c68-9fa7-dce3b86da573","resolution":{"observed_at":"2026-05-12T05:36:24.923919Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14734","last_updated":"2025-03-27T02:52:43Z","snapshot_observed_at":"2026-08-02T04:15:31.100670Z","submitted_at":"2025-03-18T21:06:21Z","title":"GR00T N1: An Open Foundation Model for Generalist Humanoid Robots","version":2},"cited_work":{"arxiv_id":"2503.14734","doi":"10.48550/arxiv.2503.14734","metadata_source":"pith","pith_arxiv_id":"2503.14734","snapshot_observed_at":"2026-07-10T23:07:47.672265Z","title":"GR00T N1: An Open Foundation Model for Generalist Humanoid Robots","venue":"cs.RO","work_id":"e2db69c7-ee8a-4cb7-a761-7b8de1dfcf97","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2503.14734","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:252fff95c75b424e51e6a6168304f4d3d2db9b3f5e4828f6ab1d917755f29eb4","observation_id":"bf4bf8f1-22f4-4bcd-b2cd-4fd870ece669","resolution":{"observed_at":"2026-05-12T05:36:24.912557Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-07-12T08:49:11.206777+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T08:49:11.206777+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"π0.5: A vision- language-action model with open-world generalization","venue":null,"work_id":"6a5c7f14-f565-4490-8d7c-9807bff4d25c","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:41dabdfe9bc9973312a511591f0c52d6f6ed8ea4305d1278145f795adae0ea97","observation_id":"86a7a6e7-2c61-4028-be60-8a39c949787a","resolution":{"observed_at":"2026-05-12T12:11:33.065834Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"π0: A vision-language-action flow model for general robot control","venue":null,"work_id":"2ade59ad-f609-44f4-b83c-b8768c47c802","year":2026},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:2c554a3ca60651d82f4f488335cf9a23898a43359a4e2abb21b2a11edd9192c8","observation_id":"f48eba0b-4134-4598-8b20-2b2756793505","resolution":{"observed_at":"2026-05-12T12:11:33.069312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":"2212.06817","doi":"10.48550/arxiv.2212.06817","metadata_source":"pith","pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-07-11T00:07:42.794081Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","venue":"cs.RO","work_id":"e11bda85-8531-46bc-a07f-d0ade3643ab1","year":2022},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:67bba9e067ad82e083d7b2f973c6d46845492a5279945f90288ba9227fbcc400","observation_id":"9fa21a8b-2f14-49c7-be3f-87dac521f47d","resolution":{"observed_at":"2026-05-12T05:36:24.888491Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:31:42.269452Z","title":"Spatialvlm: Endowing vision-language models with spatial reasoning capabilities","venue":null,"work_id":"a453f5e1-72f7-4ab7-8c7f-f38af3926115","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:bc7d6e22fd7d39320b7d6d8edfe7901b6b92c54616e4381044e2ace21bc43679","observation_id":"77d23117-d408-43d4-9d1a-9757cd1f4a01","resolution":{"observed_at":"2026-05-12T12:11:33.061909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Knowledge distillation with the reused teacher classifier","venue":null,"work_id":"52fcefd3-f702-47d0-ba62-26f5a6d23e10","year":2022},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:d428b57d7fd2b8d4b47b2a4c6867fbb2b4c6d7489a84adf3f597813ef3596ab4","observation_id":"d3f023ae-bd48-4772-a659-21a03d8fb49e","resolution":{"observed_at":"2026-05-12T12:11:33.071961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.18088","last_updated":"2025-08-27T17:52:42Z","snapshot_observed_at":"2026-08-01T01:17:47.017808Z","submitted_at":"2025-06-22T16:26:53Z","title":"RoboTwin 2.0: A Scalable Data Generator and Benchmark with Strong Domain Randomization for Robust Bimanual Robotic Manipulation","version":2},"cited_work":{"arxiv_id":"2506.18088","doi":"10.48550/arxiv.2506.18088","metadata_source":"pith","pith_arxiv_id":"2506.18088","snapshot_observed_at":"2026-07-10T23:17:45.135806Z","title":"RoboTwin 2.0: A Scalable Data Generator and Benchmark with Strong Domain Randomization for Robust Bimanual Robotic Manipulation","venue":"cs.RO","work_id":"9b985126-4a2f-4bdf-b014-2a7524ec634e","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2506.18088","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:f8636a2c8ba86889655c90da178086eaafd829f948fb66422d9187da15bc4f3b","observation_id":"a70c30b5-eb5f-4749-a389-89f8d8610c2f","resolution":{"observed_at":"2026-05-12T05:36:24.899722Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09199","last_updated":"2023-10-17T22:38:51Z","snapshot_observed_at":"2026-07-06T16:32:37.715900Z","submitted_at":"2023-10-13T15:45:19Z","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","version":2},"cited_work":{"arxiv_id":"2310.09199","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.09199","snapshot_observed_at":"2026-07-02T03:56:35.167882Z","title":"PaLi-3 vision lan- guage models: Smaller, faster, stronger","venue":null,"work_id":"d20a4352-6739-4426-8a5a-b3079cc14d00","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2310.09199","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:442c89e9a29ea302797a684c5fa1e13d5941d1f254e0727a2217830dde0ec6e9","observation_id":"52b97257-e3af-4ce2-b985-68a1af823548","resolution":{"observed_at":"2026-05-12T05:36:24.871267Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:16:04.696183Z","title":"Diffusion policy: Visuomotor policy learning via action diffusion","venue":null,"work_id":"c9c072d9-9ac9-4cb5-9f63-6a558f9b5811","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:f7195b71a2bc3c7c1f05444f0f055163b78e365974e6ec65f1cc81811a227f42","observation_id":"6bc36c34-3ce5-459c-8525-608760fe5bb3","resolution":{"observed_at":"2026-05-12T12:11:33.074517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T13:33:51.040762Z","title":"Objaverse: A universe of annotated 3d objects","venue":null,"work_id":"b22b998f-cc53-4a9a-9684-e3d0f3eebcd0","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:55de5c3dae434019b778b53d79fdf97dc62b790fc0c98269dddd11a08a2bec92","observation_id":"454005b6-74b7-4281-a547-e1eeef9b7a20","resolution":{"observed_at":"2026-05-12T12:11:33.148408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rvt: Robotic view transformer for 3d object manipulation","venue":null,"work_id":"c6ab37ab-24ad-4397-8d8c-5ffe9c6fd416","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:1ec0f356bb46a66f3176e57db2a4519d661f7329fb83536d0dc2e1ab1deda024","observation_id":"0c08331c-e246-4fa6-9091-a023d0744d34","resolution":{"observed_at":"2026-05-12T12:11:33.164911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.09619","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T16:28:38.347020Z","title":"arXiv preprint arXiv:2512.09619 (2025)","venue":null,"work_id":"dd9c7466-fb1f-484b-a3bc-874ff628b1c5","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:bfe3dc6c65b7810f59821f49a083cc71382af71462ee73787c9a39e468a00ad4","observation_id":"d886c025-cf7e-4a93-8481-fb7bd91d3c31","resolution":{"observed_at":"2026-05-12T05:31:26.193107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lora: Low-rank adaptation of large language models.Iclr, 1(2):3","venue":null,"work_id":"44a979f9-ca99-4f63-8c91-b2a309a2d218","year":2022},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:f9fc2e5b17385f5a37f3c8280c7be39927aee54a04cfc66b9e1e9fe1a4d86931","observation_id":"715b4cce-4370-4acc-9fc9-1b6d8a1b2c7a","resolution":{"observed_at":"2026-05-12T12:11:33.153632Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12871","last_updated":"2024-05-09T17:35:44Z","snapshot_observed_at":"2026-08-02T11:57:50.333488Z","submitted_at":"2023-11-18T01:21:38Z","title":"An Embodied Generalist Agent in 3D World","version":3},"cited_work":{"arxiv_id":"2311.12871","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.12871","snapshot_observed_at":"2026-07-08T02:44:27.664173Z","title":"An Embodied Generalist Agent in 3D World","venue":"cs.CV","work_id":"2c1392ae-d82d-4f8f-aef7-75a5ff0e5d2e","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2311.12871","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:6983abdce6fecec190271842e47c01baf171ba5cdabaf8ffcade08da57944a72","observation_id":"940efacc-92b9-400c-bc45-90cea4912d07","resolution":{"observed_at":"2026-05-17T14:22:18.773673Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mllms need 3d-aware representation supervision for scene understanding.arXiv e-prints, pages arXiv–2506","venue":null,"work_id":"386270ee-049b-4422-88d7-c19bb9b2cf38","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:55ce8541fbf440957ed49cd1ab64735b4544653902b950d153f36a1f2e395217","observation_id":"ba798ace-4dc1-4890-a2af-bf2dcc5632aa","resolution":{"observed_at":"2026-05-12T12:11:33.137843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"What’s “up” with vision-language models? investigating their struggle with spatial reasoning","venue":null,"work_id":"85f910d6-0b69-40f3-9159-731790bccaf4","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:df5d8e7ad9734f359f6565a428d860a1dbe947fd0466b9ddfaa2353ccfc21123","observation_id":"eab724c8-8dae-4cc4-9b6d-888b28e2c9f6","resolution":{"observed_at":"2026-05-12T12:11:33.101141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prismatic vlms: Investigating the design space of visually-conditioned language models","venue":null,"work_id":"86d3cb9f-56c7-4290-9eb5-074f0e3518e8","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:ae73781fd22508aa4ded9818d4f8067bbe5fe47d55e8263818a1ff05de9af1b0","observation_id":"7ab7566c-37b2-44ca-a372-912b59a37155","resolution":{"observed_at":"2026-05-12T12:11:33.104776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T02:27:52.773054Z","title":"3d gaussian splatting for real-time radiance field rendering.ACM Trans","venue":null,"work_id":"4a48ade7-2802-4e75-aeb9-b58be572ed7b","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:9392ab780ae0f89b4c4279808b1b88beaede71eac8ba1985daccc6b517c9b456","observation_id":"b45b0076-b0c6-4a4e-a783-2dddc83475aa","resolution":{"observed_at":"2026-05-12T12:11:33.108503Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19645","last_updated":"2025-04-28T07:49:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-27T00:30:29Z","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","version":2},"cited_work":{"arxiv_id":"2502.19645","doi":"10.48550/arxiv.2502.19645","metadata_source":"pith","pith_arxiv_id":"2502.19645","snapshot_observed_at":"2026-07-10T23:37:42.946293Z","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","venue":"cs.RO","work_id":"04f46bb3-4346-47e8-bf09-c75d91f96e87","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2502.19645","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:2dc4d34164a2a38b5ab6e78aee6add517a39c601fdd3cf1fd32b9bc176f3c51a","observation_id":"df3b2a92-fde4-4ee7-8998-fa1d832ede1a","resolution":{"observed_at":"2026-05-12T05:31:26.205311Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":"2406.09246","doi":"10.18653/v1/2022.naacl-main.68","metadata_source":"pith","pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","venue":"cs.RO","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:2e0ed4323734dfb339422cba28069c168ffc3f66ceab7191e4d1b35bbfbd14f9","observation_id":"3bea8396-6ecf-4c27-922f-6191b3b69979","resolution":{"observed_at":"2026-05-12T05:31:26.139493Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A review of robot learning for manip- ulation: Challenges, representations, and algorithms.Journal of machine learning research, 22(30):1–82","venue":null,"work_id":"a4cb7b27-42ce-4d05-b1c7-4146d4b1b69b","year":2021},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:e87f33e3e7ec915748e78abc3d0b0e51a93ce11dda31bdc8579102f32fd9bece","observation_id":"2506cb1c-e9dd-46af-bc32-65ff85fbb04c","resolution":{"observed_at":"2026-05-12T12:11:33.119351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A review of spatial reasoning and interaction for real-world robotics.Advanced Robotics, 31(5):222–242","venue":null,"work_id":"b98ac4e1-c704-4243-980e-e00c1ea0367e","year":2017},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:a90c80b9d5c8373bbe48bbf1e45a7c25c452259a0b32d2da9cde98410da337f1","observation_id":"8175d114-beaa-4b83-bc1a-5511238af15d","resolution":{"observed_at":"2026-05-12T12:11:33.123395Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T16:52:40.118211Z","title":"Pointvla: Injecting the 3d world into vision-language-action models.IEEE Robotics and Automation Letters, 11(3):2506–2513","venue":null,"work_id":"eba17e72-b20e-48f6-9e16-9c1fc6b96ebf","year":2026},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:8d50620fa45f560ad568b6b4f77490d3e74c9e0b87f02377114bc7d7e63d0581","observation_id":"e5835fe9-69b1-4615-a6eb-99a27f7d540b","resolution":{"observed_at":"2026-05-12T12:11:33.141966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.12276","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T20:16:29.394965Z","title":"Spatial forcing: Implicit spatial representation alignment for vision- language-action model","venue":null,"work_id":"0de0fc11-052f-4fee-af32-850d60b54a52","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:440f047913660566a5b5725ca8c7e8951bb51f430c1437448549ab9258f94946","observation_id":"2a8b4649-1302-4a36-8aca-c3dedbe5e7d3","resolution":{"observed_at":"2026-05-12T05:31:26.188928Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.00416","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T13:59:52.614921Z","title":"Evo-0: Vision-language-action model with implicit spatial understanding.arXiv preprint arXiv:2507.00416","venue":null,"work_id":"d1204baf-ba9c-4c49-a595-132c514f7fde","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:452bc342ccce2e66c53ce3f811a76b78330e86628b0a61084b55accb58f146e7","observation_id":"1f7d639f-3222-424c-9e96-3549df67ffae","resolution":{"observed_at":"2026-05-12T05:31:26.174886Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:16:04.690179Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36:34892–34916","venue":null,"work_id":"115823a2-8918-4227-8872-3d0a36ff07a9","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:de04f81d70af3b46113aaf41c42a9ea59f8f4928221aa47aa3d158dff22ed0a7","observation_id":"aeb4f825-3c86-4c40-a6c6-ced04b25825f","resolution":{"observed_at":"2026-05-12T12:11:33.093663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07864","last_updated":"2025-03-01T08:57:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-10T12:33:46Z","title":"RDT-1B: a Diffusion Foundation Model for Bimanual Manipulation","version":2},"cited_work":{"arxiv_id":"2410.07864","doi":"10.48550/arxiv.2410.07864","metadata_source":"pith","pith_arxiv_id":"2410.07864","snapshot_observed_at":"2026-07-10T19:47:32.618055Z","title":"RDT-1B: a Diffusion Foundation Model for Bimanual Manipulation","venue":"cs.RO","work_id":"12319725-bc7d-4c32-a229-ad270a7460bc","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2410.07864","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:083eec8389f9b202991e164c353c8fa6f40484e06c5bde59b55f650d23844436","observation_id":"dbe476cc-2772-48f6-b10c-9808912ccfa6","resolution":{"observed_at":"2026-05-12T05:31:26.219409Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":"2304.07193","doi":"10.48550/arxiv.2304.07193","metadata_source":"pith","pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-07-11T00:07:42.299741Z","title":"DINOv2: Learning Robust Visual Features without Supervision","venue":"cs.CV","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:c5ff7c03b0ef02abb98ad44b8187dde9773aef7e078fc356df4e52b16b976b1a","observation_id":"5f0b34c4-c348-44be-8e12-221c47f87d3d","resolution":{"observed_at":"2026-05-12T05:36:24.928992Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T17:15:09.637723Z","title":"Open x- embodiment: Robotic learning datasets and rt-x models: Open x-embodiment collaboration 0","venue":null,"work_id":"846c44cc-0874-4a3c-90cb-b86f68157e99","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:1de3593562991e4b56bff9e8bb43f0da3dac40ca94c579032dfd501172676d1c","observation_id":"734766e6-bd05-4d34-bfbe-5cc6ed4bc686","resolution":{"observed_at":"2026-05-12T12:11:33.145253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Film: Visual reasoning with a general conditioning layer","venue":null,"work_id":"dba07c6b-9ccf-4c9c-b835-e964a7cef234","year":2018},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:34e37f5d2acba5d4689ededb1578aa97a109bf1d6b887131001cf1a7fce6ef54","observation_id":"b1798aea-963e-4d10-9c5e-9539590e036e","resolution":{"observed_at":"2026-05-12T12:11:33.157223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15830","last_updated":"2025-05-19T02:40:18Z","snapshot_observed_at":"2026-07-06T20:26:31.558337Z","submitted_at":"2025-01-27T07:34:33Z","title":"SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model","version":5},"cited_work":{"arxiv_id":"2501.15830","doi":"10.48550/arxiv.2501.15830","metadata_source":"pith","pith_arxiv_id":"2501.15830","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model","venue":"cs.RO","work_id":"592041b3-3ca2-4836-8dd4-f8095d8a692b","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2501.15830","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:90e997ce1592ff01790d74ca3f1525c71e27f2c0f7d6c88d4ff8e33ecdc8880c","observation_id":"d5e96b2a-3d85-4d7c-a0e7-fc27ade84f9d","resolution":{"observed_at":"2026-05-12T06:12:22.898827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T09:03:29.368115Z","title":"Perceiver-actor: A multi-task transformer for robotic manipulation","venue":null,"work_id":"7bf1c712-5604-4be3-bc31-249428bc4524","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:3b4d3b9e20bae3b8305adba8dc62b61d68bfd95ce81bbe477c02093a3d375ea4","observation_id":"7304605e-475b-4f2f-ae88-4d10a8a2145c","resolution":{"observed_at":"2026-05-12T12:11:33.126906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Robospatial: Teaching spatial understanding to 2d and 3d vision-language models for robotics","venue":null,"work_id":"94881dcd-8010-4f39-8858-d4b2521ab126","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:a1a2ee9645789d9dd4bfbdeff6f385c137023ab6ccffcc4385e08115752ee111","observation_id":"df22501b-42a1-44d2-b133-cf43852717d7","resolution":{"observed_at":"2026-05-12T12:11:33.150931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.17951","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T17:38:43.949560Z","title":"Rocket: Residual-oriented multi-layer alignment for spatially- aware vision-language-action models","venue":null,"work_id":"c8e40d57-3b46-4af7-97bc-b70821035eef","year":2026},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:e23437b7e2249a506a89da637caddebb29327fa2914de28c1f1980389a9e6285","observation_id":"f5979ade-eb3a-4b41-93fe-c6b66ec3e213","resolution":{"observed_at":"2026-05-12T05:31:26.168690Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.09071","last_updated":"2025-08-13T16:47:50Z","snapshot_observed_at":"2026-07-06T22:11:52.522787Z","submitted_at":"2025-08-12T16:46:05Z","title":"GeoVLA: Empowering 3D Representations in Vision-Language-Action Models","version":2},"cited_work":{"arxiv_id":"2508.09071","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.09071","snapshot_observed_at":"2026-07-03T17:38:43.940925Z","title":"Geovla: Empowering 3d representa- tions in vision-language-action models","venue":null,"work_id":"602b66af-32d5-4baa-a248-8f0013efdf1d","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2508.09071","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:025cd0ecd00620f6d76bbb5ec4ce992ece6f0a28dde11479de92d8e7f42d7840","observation_id":"27f97e89-9f07-455e-b991-4f3b9dbc9460","resolution":{"observed_at":"2026-05-12T05:31:26.179579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12213","last_updated":"2024-05-26T19:55:26Z","snapshot_observed_at":"2026-07-06T18:16:51.116432Z","submitted_at":"2024-05-20T17:57:01Z","title":"Octo: An Open-Source Generalist Robot Policy","version":2},"cited_work":{"arxiv_id":"2405.12213","doi":"10.48550/arxiv.2405.12213","metadata_source":"pith","pith_arxiv_id":"2405.12213","snapshot_observed_at":"2026-07-10T10:07:00.871229Z","title":"Octo: An Open-Source Generalist Robot Policy","venue":"cs.RO","work_id":"f9ca0722-8855-48c3-a27a-0eefb7e19253","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2405.12213","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:869fb021e692b7f829ac84ef0c5cc3589138f6290eab9b8a21d4d1cfdc0242dc","observation_id":"553f3816-b02f-45ec-b513-2f8cf2bab091","resolution":{"observed_at":"2026-05-12T05:31:26.154335Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bridgedata v2: A dataset for robot learning at scale","venue":null,"work_id":"7e97bb70-9930-4b22-a448-35cbedf6df71","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:4ebd28599c7fcea3ecedabfdc28a992e316a55083ee747c4598535e717187826","observation_id":"57f4e4bb-af80-4afe-b382-23f7831ed237","resolution":{"observed_at":"2026-05-12T12:11:33.130223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1109/cvpr52734.2025.00499","metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T02:27:52.572720Z","title":"Vggt: Visual geometry grounded transformer","venue":"2025 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","work_id":"07a3c51b-ffe4-44d2-9239-86e84934db95","year":2025},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:65b81fcb7c91e410d7e031b59e00137c259da947a10e6c2067077167dbda3b2f","observation_id":"689ab1b1-4cb2-410b-b924-eac27ae39780","resolution":{"observed_at":"2026-05-12T12:11:33.090295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-07-20T18:21:56.85888+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-20T18:21:56.85888+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Depth anything: Unleashing the power of large-scale unlabeled data","venue":null,"work_id":"897588a9-9451-4bc7-82fe-c3a1d974be48","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:d0e6289b75308986a399be1aa17b7ca365b6db23943b7ccf09fa853ae5cb416a","observation_id":"482c5687-db62-43d7-8405-4cb4c7ae365d","resolution":{"observed_at":"2026-05-12T12:11:33.096839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Depth anything v2.Advances in Neural Information Processing Systems, 37:21875– 21911","venue":null,"work_id":"80736852-1e8d-46cd-ad21-fb1289bc85e5","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:1d6f4e83bb6a65c9fb0f48eb16c8a36c0b3250ffbb4993890861473bd9cc09ea","observation_id":"903892b5-6838-4477-a974-7a5381729265","resolution":{"observed_at":"2026-05-12T12:11:33.112063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scannet++: A high- fidelity dataset of 3d indoor scenes","venue":null,"work_id":"c5b75392-0e50-42be-b6e4-d460054584b6","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:9bee01298ec18f74dbc0d2d0b5c15ff205452e5999aff670c6725ed2ea7cf8c7","observation_id":"bbe0a506-6e8f-4564-9f9f-dce213bb1c31","resolution":{"observed_at":"2026-05-12T12:11:33.133373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10721","last_updated":"2024-06-15T19:22:51Z","snapshot_observed_at":"2026-07-06T18:31:36.504766Z","submitted_at":"2024-06-15T19:22:51Z","title":"RoboPoint: A Vision-Language Model for Spatial Affordance Prediction for Robotics","version":1},"cited_work":{"arxiv_id":"2406.10721","doi":"10.48550/arxiv.2406.10721","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.10721","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Robopoint: A vision-language model for spatial affordance prediction for robotics","venue":null,"work_id":"af457b2b-ad31-4178-ae3c-b18aaa2623ec","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2406.10721","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:def0f82ca02d716fdac7ac48ad95b067ef54cabfb49f35f55e44770a665f6fa4","observation_id":"97c48c6d-4df8-4c1a-86fd-39b946f8bd03","resolution":{"observed_at":"2026-05-12T05:31:26.143646Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improving 2d feature representations by 3d-aware fine-tuning","venue":null,"work_id":"39b7f73d-d8bc-4925-a8ff-5bd9a274e805","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:73cfd91ec464bcb36ebff9f6d3114fdd2f0409b8661e168c3f7afb50e306e787","observation_id":"18578f66-026b-4a86-9dd7-05bb52f171a7","resolution":{"observed_at":"2026-05-12T12:11:33.115308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T16:12:38.217835Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"7d49d7f8-2cb8-4e91-8b78-0753271fb6a2","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:1927e860c2da5de27203e82ba0b67a452586a58c591bbf4c7cd3644edbeb7c96","observation_id":"9b5d40de-7725-4c42-a780-9c37402bdccf","resolution":{"observed_at":"2026-05-12T12:11:33.159779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09631","last_updated":"2024-03-14T17:58:41Z","snapshot_observed_at":"2026-07-06T17:44:45.924645Z","submitted_at":"2024-03-14T17:58:41Z","title":"3D-VLA: A 3D Vision-Language-Action Generative World Model","version":1},"cited_work":{"arxiv_id":"2403.09631","doi":"10.48550/arxiv.2403.09631","metadata_source":"pith","pith_arxiv_id":"2403.09631","snapshot_observed_at":"2026-07-10T07:26:54.418557Z","title":"3D-VLA: A 3D Vision-Language-Action Generative World Model","venue":"cs.CV","work_id":"aebf924c-e761-437e-9cee-f1ccc2e427bd","year":2024},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"cited_paper":"/paper/2403.09631","citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:f9e763a5d8c05702ada8987d54bf8618d1c1fc3ac0fc0767c6290309fca411d3","observation_id":"d462be3f-9b37-4448-b91f-275895efa924","resolution":{"observed_at":"2026-05-13T18:18:27.368966Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","venue":null,"work_id":"d98b0e75-1374-4b1b-8108-7fe60ca97030","year":2023},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:777735d820a62cd2d9d44489ec8c0b2c72c5c883302cd20ea45b5b27ad0e7c9b","observation_id":"a8584f01-c829-465d-920b-0c1b98325f7b","resolution":{"observed_at":"2026-05-12T12:11:33.162351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This task requires precise spatial perception to locate the screen and hinge, as well as smooth and controlled motion to avoid damaging the articulated structure during contact","venue":null,"work_id":"d356f341-7829-401b-8f4e-0e03f29f4fb9","year":null},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:22eb4650d973b30f945e8b5fefa2f56d2bcbec9a29d5d52cef1e113ebe059346","observation_id":"ae111cc0-b8df-4cbf-b2d2-735e49b36e8f","resolution":{"observed_at":"2026-05-12T12:11:33.083821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This task requires accurate object localization and a smooth transfer trajectory to ensure stable grasping and precise placement without dropping the object","venue":null,"work_id":"ea064b33-24eb-4679-96c9-6bee21a961ad","year":null},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:928448ddb4c74ca92f45621a2e6fe4ca0f83de29b9839072e4b57cdaa9434ce3","observation_id":"f9c5205d-ad76-4eef-adde-3a561af8b304","resolution":{"observed_at":"2026-05-12T12:11:33.087208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e3177915-2dd1-4554-82b7-f509926f540d","year":null},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:0aaf5e48bec1d13c26b615a13e65360273b48486458b30632688e5422aabd82f","observation_id":"c17223a3-08d7-41d7-842e-a7adb38e5216","resolution":{"observed_at":"2026-05-12T12:11:33.080104Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c7858d9f-8ed5-4c47-9544-d1ecb5f15268","year":null},"citing_paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-12T05:09:21.028373Z"},"links":{"citing_paper":"/paper/2605.10485"},"observation_digest":"sha256:79ab527dea95e07e137555ea69606b3315012db6d0f88d3ba7c61ff0ceccbee9","observation_id":"556c2877-d691-4ec2-a8da-7b4e56801933","resolution":{"observed_at":"2026-05-12T12:11:33.077238Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.10485","last_updated":"2026-05-11T12:44:26Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-02T06:15:05.626635Z","submitted_at":"2026-05-11T12:44:26Z","title":"VEGA: Visual Encoder Grounding Alignment for Spatially-Aware Vision-Language-Action Models"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":2,"verified_exact":19,"verified_fuzzy":30},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"thesis":"As of 3 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 1 inbound Pith citation observation for arXiv:2605.10485."}