{"as_of":"2026-08-07T15:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d12790ecb32119326eca480a4782999e4b6267d08ff75c77032c85646cfde9e7","coverage":[{"denominator":137,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:57:29.897079Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.12374/citation-record","integrity":"/paper/2506.12374/integrity","json":"/paper/2506.12374/citation-record.json","paper":"/paper/2506.12374"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:25.081678Z","title":"Flamingo: a visual language model for few-shot learning.Advances in Neural Information Processing Systems, 35:23716–23736, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.081678Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:f44b9383d0bb777bd9184fbdf56128077cb541af3b884c98ada3b9dcb0a4c3e4","observation_id":"45199b3e-0170-40a6-bbc6-d428e817f69f","resolution":{"observed_at":"2026-08-07T00:57:25.081678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:25.150439Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.150439Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:9163190429f7fcbcd1659d46cbcda583b3d3584982275c562d4f63dcae7648ad","observation_id":"e90b2662-f6d1-43e4-b255-88be81a0910b","resolution":{"observed_at":"2026-08-07T00:57:25.150439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:25.223630Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.223630Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:18a9876f8371ef854421e4ad0438543f813465c458a5bc5fb9d0556a857f19fc","observation_id":"a76572f7-59b4-49f5-a18c-98f5025b2ad7","resolution":{"observed_at":"2026-08-07T00:57:25.223630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:25.300306Z","title":"Zero-shot text-to-image generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.300306Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:5a26ab302ae49b43512028ca3116125317d8c72132d057e8c3ce9a6e7df05b98","observation_id":"4ba83d3d-98be-424e-8a83-be2de65f9adf","resolution":{"observed_at":"2026-08-07T00:57:25.300306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03378","last_updated":"2023-03-06T18:58:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-06T18:58:06Z","title":"PaLM-E: An Embodied Multimodal Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03378","snapshot_observed_at":"2026-08-07T00:57:25.396676Z","title":"Palm-e: An embodied multimodal language model.arXiv preprint arXiv:2303.03378, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.396676Z"},"links":{"cited_paper":"/paper/2303.03378","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:b71d2eeac9b03aa55d03035601af00af24502749fc5ff48b4849d847b64b3c4b","observation_id":"8edbc6a0-d9e7-4f3f-a35a-387f5a3a0de7","resolution":{"observed_at":"2026-08-07T00:57:25.396676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T00:57:25.479391Z","title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond.arXiv preprint arXiv:2308.12966, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.479391Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:e902e1622795f55f82a5accd3cfef2fc453bdb7b4eddea8255e0ff23cf5cd9b4","observation_id":"6518237b-bf54-4ede-81e0-f66736cc7e99","resolution":{"observed_at":"2026-08-07T00:57:25.479391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T00:57:25.576785Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.576785Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:30d95b49dea1e1b1db676114c2d0f10c6320d739e2e423eed9f8069ed5c3cac0","observation_id":"b8a27dc8-e4ae-4833-92ba-aa3a621e239b","resolution":{"observed_at":"2026-08-07T00:57:25.576785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.01691","last_updated":"2022-08-16T16:06:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-04T17:57:11Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.01691","snapshot_observed_at":"2026-08-07T00:57:25.641202Z","title":"Do as i can, not as i say: Grounding language in robotic affordances.arXiv preprint arXiv:2204.01691, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.641202Z"},"links":{"cited_paper":"/paper/2204.01691","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:3b4660b07f13707cb2eb676e5f69fd83f8c8286569224a1053c17b5a14e59822","observation_id":"c8333fc0-7cb3-41f0-8bf5-79688f81c4bd","resolution":{"observed_at":"2026-08-07T00:57:25.641202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:25.750801Z","title":"Open-vocabulary queryable scene representations for real world planning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.750801Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:71ae97b895e049777561674597903ecec5c4e4ef94c327106d36be0c1962f474","observation_id":"cbf9ea69-12ff-4097-a862-f5a33c4881ad","resolution":{"observed_at":"2026-08-07T00:57:25.750801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:25.851964Z","title":"Lm-nav: Robotic navigation with large pre-trained models of language, vision, and action","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.851964Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:7036deaf77f4b6c306e6061f2b1e44c7228eeda876b7b465789ceab4321d0b4f","observation_id":"b638acb7-e6f4-41bc-bb7b-a8ce4b092d30","resolution":{"observed_at":"2026-08-07T00:57:25.851964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.00598","last_updated":"2022-05-27T17:52:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-01T17:43:13Z","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.00598","snapshot_observed_at":"2026-08-07T00:57:25.947466Z","title":"So- cratic models: Composing zero-shot multimodal reasoning with language.arXiv preprint arXiv:2204.00598, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:25.947466Z"},"links":{"cited_paper":"/paper/2204.00598","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:e48ed1635b75fd91a1a191f3d3ac61b48bb6f5901f7a9c0c49a6f30ad7de9499","observation_id":"1fd7d921-6d8c-46a1-8028-a7e487bca854","resolution":{"observed_at":"2026-08-07T00:57:25.947466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:26.062840Z","title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.062840Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:998e5730d531f8a45f657701ef58592921a778769a5a0f6f78d3b084dfe17f8a","observation_id":"0d1561d3-4d7c-4800-a5ac-e1472a3bf328","resolution":{"observed_at":"2026-08-07T00:57:26.062840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:26.188599Z","title":"Code as policies: Language model programs for embodied control","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.188599Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:e8b5eb829f9175b15dbc49e32427dc7a35d864e23ed5986c8bab9f3d5ba0f7a3","observation_id":"d401c5fd-abdf-45e9-9f76-541046427ecf","resolution":{"observed_at":"2026-08-07T00:57:26.188599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.05973","last_updated":"2023-11-02T06:53:37Z","snapshot_observed_at":"2026-08-05T01:03:23.456778Z","submitted_at":"2023-07-12T07:40:48Z","title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.05973","snapshot_observed_at":"2026-08-07T00:57:26.264561Z","title":"V oxposer: Composable 3d value maps for robotic manipulation with language models.arXiv preprint arXiv:2307.05973, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.264561Z"},"links":{"cited_paper":"/paper/2307.05973","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:3391d2ba30cde704a6eb715acb692c7e1f8aa9ab1367d1599fda59ce3ab9a9b4","observation_id":"3ee45808-8a8d-4e6c-a6c0-1b2f11779223","resolution":{"observed_at":"2026-08-07T00:57:26.264561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07872","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-02T06:03:03.152035Z","submitted_at":"2024-02-12T18:33:47Z","title":"PIVOT: Iterative Visual Prompting Elicits Actionable Knowledge for VLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07872","snapshot_observed_at":"2026-08-07T00:57:26.306305Z","title":"Pivot: Iterative visual prompting elicits actionable knowledge for vlms.arXiv preprint arXiv:2402.07872, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.306305Z"},"links":{"cited_paper":"/paper/2402.07872","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:3783ff480669186b709e921cdcb145f196cd6b7eda0ec81f177a9c6bcd9f3be2","observation_id":"f898e174-1805-4323-9a7f-b500b241d761","resolution":{"observed_at":"2026-08-07T00:57:26.306305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01652","last_updated":"2024-11-12T04:33:26Z","snapshot_observed_at":"2026-07-29T23:37:00.228879Z","submitted_at":"2024-09-03T06:45:22Z","title":"ReKep: Spatio-Temporal Reasoning of Relational Keypoint Constraints for Robotic Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01652","snapshot_observed_at":"2026-08-07T00:57:26.372509Z","title":"Rekep: Spatio- temporal reasoning of relational keypoint constraints for robotic manipulation.arXiv preprint arXiv:2409.01652, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.372509Z"},"links":{"cited_paper":"/paper/2409.01652","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:8ee50134c6861c8e25477651b0865316c9cc97a09375a02ab5796447e521ab89","observation_id":"f1494ef9-d559-40ea-9c75-588bc06f24d5","resolution":{"observed_at":"2026-08-07T00:57:26.372509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01616","last_updated":"2025-03-03T14:49:52Z","snapshot_observed_at":"2026-07-06T20:45:46.299882Z","submitted_at":"2025-03-03T14:49:52Z","title":"RoboDexVLM: Visual Language Model-Enabled Task Planning and Motion Control for Dexterous Robot Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01616","snapshot_observed_at":"2026-08-07T00:57:26.482172Z","title":"Robodexvlm: Visual language model-enabled task planning and motion control for dexterous robot manipu- lation.arXiv preprint arXiv:2503.01616, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.482172Z"},"links":{"cited_paper":"/paper/2503.01616","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:7b69c33ba597fe09bc07ff27c613c1580f478856a895868f40f21e2b2e0b9188","observation_id":"7699befe-e1bc-4ead-880f-0e1beb5288f5","resolution":{"observed_at":"2026-08-07T00:57:26.482172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.21530","last_updated":"2025-04-30T11:26:40Z","snapshot_observed_at":"2026-07-06T21:17:00.913307Z","submitted_at":"2025-04-30T11:26:40Z","title":"RoboGround: Robotic Manipulation with Grounded Vision-Language Priors","version":1},"cited_work":{"arxiv_id":"2504.21530","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.21530","snapshot_observed_at":"2026-08-07T00:57:31.251745Z","title":"RoboGround: Robotic Manipulation with Grounded Vision-Language Priors","venue":"cs.RO","work_id":"d91941a4-8272-4c55-9820-b9dceb66320a","year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.612635Z"},"links":{"cited_paper":"/paper/2504.21530","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:be0fd6b473a5347560aecbe6b5c8459a57e8705010a7c86b61f807f71b3baf48","observation_id":"7c29c10a-ade1-4c72-91e6-45a7d04c8a88","resolution":{"observed_at":"2026-08-07T00:57:31.256067Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:26.751072Z","title":"Llm-grounder: Open-vocabulary 3d visual grounding with large lan- guage model as an agent","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.751072Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:dbb163434ffff7ee0a5ec24dc82a9dc35bd9dfe4fbd32b7f94314b45eaf74215","observation_id":"ba1a1b09-4043-4a96-8a6a-f2fe481d4284","resolution":{"observed_at":"2026-08-07T00:57:26.751072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13438","last_updated":"2024-10-30T00:47:52Z","snapshot_observed_at":"2026-07-06T17:47:34.522070Z","submitted_at":"2024-03-18T17:38:29Z","title":"SpatialPIN: Enhancing Spatial Reasoning Capabilities of Vision-Language Models through Prompting and Interacting 3D Priors","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13438","snapshot_observed_at":"2026-08-07T00:57:26.863014Z","title":"Spatialpin: Enhancing spatial reasoning capabilities of vision-language models through prompting and interacting 3d priors.arXiv preprint arXiv:2403.13438, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:26.863014Z"},"links":{"cited_paper":"/paper/2403.13438","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:3832c6dae1e1908f3915dd2725dbece6545ecb65ff0aad38262dd15b41b21a48","observation_id":"4a81f21e-71f4-46d6-a2cb-b4dd1f8683bb","resolution":{"observed_at":"2026-08-07T00:57:26.863014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:27.025972Z","title":"Spatialvlm: Endowing vision-language models with spatial reasoning capabilities","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.025972Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:65dea0e7c59c3695c0fed05d3287c344d079474efa74ea203d2817a2f2e3285c","observation_id":"0a19d958-9ac3-4651-8fac-421e7d39b69b","resolution":{"observed_at":"2026-08-07T00:57:27.025972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01584","last_updated":"2024-10-15T01:16:20Z","snapshot_observed_at":"2026-07-06T18:24:38.958815Z","submitted_at":"2024-06-03T17:59:06Z","title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01584","snapshot_observed_at":"2026-08-07T00:57:27.127004Z","title":"Spatialrgpt: Grounded spatial reasoning in vision language models.arXiv preprint arXiv:2406.01584, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.127004Z"},"links":{"cited_paper":"/paper/2406.01584","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:c19f959a9046a215105cdafc26530194f9b5e13c171eead3471ec023525b845f","observation_id":"e85307f9-bf8d-42ae-aaf8-e44fdc65d69a","resolution":{"observed_at":"2026-08-07T00:57:27.127004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15485","last_updated":"2025-08-13T18:39:29Z","snapshot_observed_at":"2026-07-06T21:12:46.642950Z","submitted_at":"2025-04-21T23:38:43Z","title":"CAPTURe: Evaluating Spatial Reasoning in Vision Language Models via Occluded Object Counting","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15485","snapshot_observed_at":"2026-08-07T00:57:27.167072Z","title":"Capture: Evaluating spatial reasoning in vision language models via occluded object counting.arXiv preprint arXiv:2504.15485, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.167072Z"},"links":{"cited_paper":"/paper/2504.15485","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:97b504f259325ed9f53478a0bbd90175fef89fa38d9c54f166841cbc7a85c109","observation_id":"89ed4db9-1ac1-465e-9632-203f668c3226","resolution":{"observed_at":"2026-08-07T00:57:27.167072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-07T07:37:52.384183Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-07T00:57:27.247850Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.247850Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:cd5b136a39e769bd00fe5584badd37b38018a466ce42be0c28b8cbff1cb4e458","observation_id":"6e20543d-3d2d-4381-8c77-b14dd9b70932","resolution":{"observed_at":"2026-08-07T00:57:27.247850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:27.372595Z","title":"Zero-shot visual reasoning by vision- language models: Benchmarking and analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.372595Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:8b5e7a99946bd7c9061a77799ad539e5934acc310b2ea406de38f6bf94068333","observation_id":"af814dc3-26b1-45ef-83e8-940e705c8bfa","resolution":{"observed_at":"2026-08-07T00:57:27.372595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:27.459312Z","title":"How to enable llm with 3d capacity? a survey of spatial reasoning in llm.arXiv preprint arXiv:2504.05786, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.459312Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:22b27b12bf3c84924b735525fc2f3f3572eae01bcfb31e1499163adfe8683d7a","observation_id":"7b350e45-f5a0-40d5-a504-12a5c2a2c2bc","resolution":{"observed_at":"2026-08-07T00:57:27.459312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-07T00:57:27.528961Z","title":"Thinking in space: How multimodal large language models see, remember, and recall spaces.arXiv preprint arXiv:2412.14171, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.528961Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:95d4198558b89e594c5df7c84f4daea9d16f3e2bc115a1285080b4f2a93ffc44","observation_id":"d6bfe043-b776-4725-ac7b-a1fba9f5b982","resolution":{"observed_at":"2026-08-07T00:57:27.528961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18125","last_updated":"2025-04-27T06:50:23Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:59:11Z","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.18125","snapshot_observed_at":"2026-08-07T00:57:27.599419Z","title":"Llava-3d: A simple yet effective pathway to empowering lmms with 3d-awareness.arXiv preprint arXiv:2409.18125, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.599419Z"},"links":{"cited_paper":"/paper/2409.18125","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:7f3c39b2238255ea0f34b94bfe87733b7604e52bd3afa95af2470bbe89f4486b","observation_id":"c8499945-4b1d-4d4d-89e3-eaac8adcd815","resolution":{"observed_at":"2026-08-07T00:57:27.599419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:27.720163Z","title":"Agent3d-zero: An agent for zero-shot 3d understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.720163Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:f49b48b16251124b55577c0dc62871b7f399deff10200c67ebf7903971eb108b","observation_id":"06b3bc4a-023e-42d1-9676-72c568e0c030","resolution":{"observed_at":"2026-08-07T00:57:27.720163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:27.842050Z","title":"Shapellm: Universal 3d object understanding for embodied interaction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.842050Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:cb019a1e27d61ba38af81e7cae01c2908d2a79437f588b24b218409f7f8160e3","observation_id":"c990ba0e-9bf6-41c4-9c72-612f6e2a9d5f","resolution":{"observed_at":"2026-08-07T00:57:27.842050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11401","last_updated":"2024-03-22T18:52:51Z","snapshot_observed_at":"2026-08-06T11:39:56.281668Z","submitted_at":"2024-03-18T01:18:48Z","title":"Scene-LLM: Extending Language Model for 3D Visual Understanding and Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11401","snapshot_observed_at":"2026-08-07T00:57:28.005755Z","title":"Scene-llm: Extending language model for 3d visual understanding and reasoning.arXiv preprint arXiv:2403.11401, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.005755Z"},"links":{"cited_paper":"/paper/2403.11401","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:714a0dd7abc8b09b1ca98a517a8b39c8b45a2a3eeb0ae846c02b5ae5d814d491","observation_id":"d451d68d-d5ea-49c2-ac94-441f12a5004d","resolution":{"observed_at":"2026-08-07T00:57:28.005755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16198","last_updated":"2024-07-23T06:02:30Z","snapshot_observed_at":"2026-07-06T18:50:31.528338Z","submitted_at":"2024-07-23T06:02:30Z","title":"INF-LLaVA: Dual-perspective Perception for High-Resolution Multimodal Large Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16198","snapshot_observed_at":"2026-08-07T00:57:28.102223Z","title":"Inf-llava: Dual-perspective perception for high-resolution multimodal large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.102223Z"},"links":{"cited_paper":"/paper/2407.16198","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:b98f53b4bc1d48576238bd8b07ebea4cd40854299e184298ff942a90db9ca022","observation_id":"e5a6f8f5-13d6-4362-af1f-7a0192560783","resolution":{"observed_at":"2026-08-07T00:57:28.102223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11370","last_updated":"2025-08-20T15:45:11Z","snapshot_observed_at":"2026-08-02T03:18:39.824103Z","submitted_at":"2023-12-18T17:36:20Z","title":"G-LLaVA: Solving Geometric Problem with Multi-Modal Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11370","snapshot_observed_at":"2026-08-07T00:57:28.210537Z","title":"G-llava: Solving geometric problem with multi-modal large language model.arXiv preprint arXiv:2312.11370, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.210537Z"},"links":{"cited_paper":"/paper/2312.11370","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:e7d018d8e3972a3137b840888d5ae68b4d68edb312396e67bd9dce1432f51099","observation_id":"7764fc4b-8cf7-4717-926f-f3a11a814c8d","resolution":{"observed_at":"2026-08-07T00:57:28.210537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:28.331337Z","title":"Manipvqa: Injecting robotic affordance and physically grounded information into multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.331337Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:f90f834ee6ca20a008083aff989d8821439ddf334a1e1a93712bfa28c4e93441","observation_id":"3c265a4b-d4ec-410e-bbe8-c6540ef8450c","resolution":{"observed_at":"2026-08-07T00:57:28.331337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:28.492183Z","title":"Robovqa: Multimodal long-horizon reasoning for robotics","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.492183Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:ba9b188392085445c83f26a4ad0072e2f5afa59a2b119777b6b92ad5a6f455e6","observation_id":"1ceeb23e-fa10-453e-b8d1-9f5377370db9","resolution":{"observed_at":"2026-08-07T00:57:28.492183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2003.04641","last_updated":"2023-02-21T05:20:30Z","snapshot_observed_at":"2026-07-06T09:03:37.435861Z","submitted_at":"2020-03-10T11:30:09Z","title":"MQA: Answering the Question via Robotic Manipulation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.04641","snapshot_observed_at":"2026-08-07T00:57:28.618914Z","title":"Mqa: Answering the question via robotic manipulation.arXiv preprint arXiv:2003.04641, 2020","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.618914Z"},"links":{"cited_paper":"/paper/2003.04641","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:0cf57d0f6c1ca4bbd44532c085e91df29a0bb609c0fc3d1cd0e902f8ce8b1499","observation_id":"b4f339b0-0fb5-4577-9c16-10da8048e90a","resolution":{"observed_at":"2026-08-07T00:57:28.618914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:28.668841Z","title":"Robotvqa—a scene-graph-and deep-learning-based visual question answering system for robot manipulation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.668841Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:f204626bc490126e40550d8b125133165e4047e5c5063405a7cf8383b2f32351","observation_id":"d2251920-b18c-4576-b55b-d1fe88529cf5","resolution":{"observed_at":"2026-08-07T00:57:28.668841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:28.748826Z","title":"A visual questioning answering approach to enhance robot localization in indoor environments.Frontiers in Neurorobotics, 17:1290584, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.748826Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:f53e176c4d02b73545974f08261b731aa7721fe490ec21f6560e08a09fedd95f","observation_id":"f7b5ad7a-a19a-4970-88a9-af407adeb473","resolution":{"observed_at":"2026-08-07T00:57:28.748826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.09829","last_updated":"2024-07-13T09:42:02Z","snapshot_observed_at":"2026-08-02T11:24:42.708982Z","submitted_at":"2024-07-13T09:42:02Z","title":"VLMPC: Vision-Language Model Predictive Control for Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.09829","snapshot_observed_at":"2026-08-07T00:57:28.855409Z","title":"Vlmpc: Vision-language model predictive control for robotic manipulation.arXiv preprint arXiv:2407.09829, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.855409Z"},"links":{"cited_paper":"/paper/2407.09829","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:a1ca2cef409d9f7ff1ebc67ed43cc4d2483d69d8c64742acc1c13525f1870940","observation_id":"007edbd3-b00c-48af-8651-72a8d2f5f8a5","resolution":{"observed_at":"2026-08-07T00:57:28.855409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16707","last_updated":"2025-02-23T20:42:15Z","snapshot_observed_at":"2026-08-07T01:24:33.202434Z","submitted_at":"2025-02-23T20:42:15Z","title":"Reflective Planning: Vision-Language Models for Multi-Stage Long-Horizon Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.16707","snapshot_observed_at":"2026-08-07T00:57:28.983421Z","title":"Re- flective planning: Vision-language models for multi-stage long-horizon robotic manipulation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:28.983421Z"},"links":{"cited_paper":"/paper/2502.16707","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:c2e75b978825177496c8f5f4a417fa6053233bee753b31b2df4e492d7cbd2835","observation_id":"02853fee-6955-4fdc-a7a1-b06257ea453f","resolution":{"observed_at":"2026-08-07T00:57:28.983421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.049705Z","title":"Open-world task and motion planning via vision-language model inferred constraints.arXiv preprint arXiv:2411.08253, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.049705Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:3fc2d29419b8210e0b43073a767a326eea7aecbd3d3305150bfd522224490798","observation_id":"883cd5eb-1dda-4aad-99f3-c20eb39e9c0c","resolution":{"observed_at":"2026-08-07T00:57:29.049705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-07T00:57:29.064263Z","title":"Rt-2: Vision-language- action models transfer web knowledge to robotic control.arXiv preprint arXiv:2307.15818, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.064263Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:643c225cc1eb9bc046c899ba0439bfd8791514e81a6f6b9458c621867f69cb12","observation_id":"5343a5cc-08c4-407a-a1ba-15e190d79ea4","resolution":{"observed_at":"2026-08-07T00:57:29.064263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-07T00:57:29.147844Z","title":"Rt-1: Robotics transformer for real-world control at scale.arXiv preprint arXiv:2212.06817, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.147844Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:78e48c9630600d41b592148b92e70f9c897af0390d7ccf1d9c94c9b6673965c2","observation_id":"4e079adf-50b6-4d9a-8d6b-0816a3c10146","resolution":{"observed_at":"2026-08-07T00:57:29.147844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04339","last_updated":"2024-12-14T18:41:03Z","snapshot_observed_at":"2026-08-06T21:10:19.901649Z","submitted_at":"2024-06-06T17:59:47Z","title":"RoboMamba: Efficient Vision-Language-Action Model for Robotic Reasoning and Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04339","snapshot_observed_at":"2026-08-07T00:57:29.206479Z","title":"Robomamba: Multimodal state space model for efficient robot reasoning and manipulation.arXiv preprint arXiv:2406.04339, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.206479Z"},"links":{"cited_paper":"/paper/2406.04339","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:8c2527a5ae1a5360cd0cf422a198253ae4ba22efd54e496a15423428fffb5798","observation_id":"38cca183-d0f7-48ce-8f4d-94170850d9d4","resolution":{"observed_at":"2026-08-07T00:57:29.206479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.283597Z","title":"Scaling proprioceptive-visual learn- ing with heterogeneous pre-trained transformers.Advances in Neural Information Processing Systems, 37:124420–124450, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.283597Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:d7c2201d37ea2573a89af0dca0e7dfed7f415fb00537ba3b36b03391b1335f92","observation_id":"6508197f-7505-44fb-ab49-954eae7e8b86","resolution":{"observed_at":"2026-08-07T00:57:29.283597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.13705","last_updated":"2023-04-23T19:10:53Z","snapshot_observed_at":"2026-08-03T01:22:01.078078Z","submitted_at":"2023-04-23T19:10:53Z","title":"Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.13705","snapshot_observed_at":"2026-08-07T00:57:29.343756Z","title":"Learning fine-grained bimanual manipulation with low-cost hardware.arXiv preprint arXiv:2304.13705, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.343756Z"},"links":{"cited_paper":"/paper/2304.13705","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:e93e4aa8d8c5fbdb9001d2651dece7b819360e38a324d03c4bc7b598c8244d2d","observation_id":"cda072aa-ae8e-446f-91f5-afd6562b6005","resolution":{"observed_at":"2026-08-07T00:57:29.343756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.02117","last_updated":"2024-01-04T07:55:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-04T07:55:53Z","title":"Mobile ALOHA: Learning Bimanual Mobile Manipulation with Low-Cost Whole-Body Teleoperation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.02117","snapshot_observed_at":"2026-08-07T00:57:29.426099Z","title":"Mobile aloha: Learning bimanual mobile manipulation with low-cost whole-body teleoperation.arXiv preprint arXiv:2401.02117, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.426099Z"},"links":{"cited_paper":"/paper/2401.02117","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:107d727c1b39aa43d7ebdb4a0bee7f1b1b33e414be9a2700e61d1fbfd51ba42c","observation_id":"56a818c4-2a61-47a4-9e76-3aa21b770964","resolution":{"observed_at":"2026-08-07T00:57:29.426099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-07T00:57:29.508963Z","title":"Openvla: An open-source vision-language-action model.arXiv preprint arXiv:2406.09246, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.508963Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:29942e4f169fa4d77188cea636ee4bb5284022905295d31fce73c0a2ad4a3c3e","observation_id":"ba9b3811-0d51-4698-b11e-46816b7ede77","resolution":{"observed_at":"2026-08-07T00:57:29.508963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07864","last_updated":"2025-03-01T08:57:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-10T12:33:46Z","title":"RDT-1B: a Diffusion Foundation Model for Bimanual Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07864","snapshot_observed_at":"2026-08-07T00:57:29.572917Z","title":"Rdt-1b: a diffusion foundation model for bimanual manipulation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.572917Z"},"links":{"cited_paper":"/paper/2410.07864","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:ed1b0b8c83e3e4f4d75a5b051787ce05280d566005d6dbd7eb38cfbcb6a54a38","observation_id":"960a02d4-e7c6-4bec-8684-ef955625343a","resolution":{"observed_at":"2026-08-07T00:57:29.572917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03293","last_updated":"2025-06-04T08:30:06Z","snapshot_observed_at":"2026-07-06T20:01:32.313250Z","submitted_at":"2024-12-04T13:11:38Z","title":"Diffusion-VLA: Generalizable and Interpretable Robot Foundation Model via Self-Generated Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03293","snapshot_observed_at":"2026-08-07T00:57:29.614110Z","title":"Diffusion-vla: Scaling robot foundation models via unified diffusion and autoregression.arXiv preprint arXiv:2412.03293, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.614110Z"},"links":{"cited_paper":"/paper/2412.03293","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:9a5af1f368e4c950f0c34e6be843afd1504508767d93916bbaee7a48968b68ce","observation_id":"3c9762e3-034f-4bc4-9d83-b3e51d650d2b","resolution":{"observed_at":"2026-08-07T00:57:29.614110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-07T00:57:29.690756Z","title":"π0: A vision-language-action flow model for general robot control.arXiv preprint arXiv:2410.24164, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.690756Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:81ec5e8f460fd77c0c1d788d6bd3008b6acb7297c18e2e9e8ce2186ae73ef4d5","observation_id":"0be1e207-5932-42b9-b829-029e1581e81e","resolution":{"observed_at":"2026-08-07T00:57:29.690756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16054","last_updated":"2025-04-22T17:31:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:31:29Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16054","snapshot_observed_at":"2026-08-07T00:57:29.709441Z","title":"π0.5: A vision- language-action model with open-world generalization.arXiv preprint arXiv:2504.16054, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.709441Z"},"links":{"cited_paper":"/paper/2504.16054","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:e1267b40ce59205a66a6752eb69cb7f25820b913fc1c49494b7e1f6956d63920","observation_id":"9a508b3a-2b41-4f7d-8f98-022b07172a94","resolution":{"observed_at":"2026-08-07T00:57:29.709441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18915","last_updated":"2024-08-29T16:07:30Z","snapshot_observed_at":"2026-07-06T18:37:43.309281Z","submitted_at":"2024-06-27T06:12:01Z","title":"Manipulate-Anything: Automating Real-World Robots using Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.18915","snapshot_observed_at":"2026-08-07T00:57:29.714057Z","title":"Manipulate-anything: Automating real-world robots using vision-language models.arXiv preprint arXiv:2406.18915, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.714057Z"},"links":{"cited_paper":"/paper/2406.18915","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:9df23df8d0f3c3d72d4b9c825f30c55b4be38276d163e72148b55450d4ac6561","observation_id":"44cfb571-0f28-4941-8826-e53bf5b252d0","resolution":{"observed_at":"2026-08-07T00:57:29.714057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.718582Z","title":"Skillman—a skill-based robotic manipulation framework based on perception and reasoning.Robotics and Autonomous Systems, 134:103653, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.718582Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:c726a5adf08967afb1ca793e4dda9fc7a53df112f0502803b5ff12a3a8de83da","observation_id":"d63ef4ae-4dc7-4a58-9977-ef06d5db41d7","resolution":{"observed_at":"2026-08-07T00:57:29.718582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08643","last_updated":"2025-02-18T16:45:59Z","snapshot_observed_at":"2026-08-03T17:58:18.077660Z","submitted_at":"2025-02-12T18:57:22Z","title":"A Real-to-Sim-to-Real Approach to Robotic Manipulation with VLM-Generated Iterative Keypoint Rewards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08643","snapshot_observed_at":"2026-08-07T00:57:29.723051Z","title":"A real-to-sim-to-real approach to robotic manipulation with vlm-generated iterative keypoint rewards.arXiv preprint arXiv:2502.08643, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.723051Z"},"links":{"cited_paper":"/paper/2502.08643","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:2e9c73abe34b3bcc3f1da7b17b5bfe339ed15c5c68c4068402ebf65bc71f15b0","observation_id":"680fdbc5-9dcc-42c6-94f7-1772805f9740","resolution":{"observed_at":"2026-08-07T00:57:29.723051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.727266Z","title":"Sofar: Language-grounded orientation bridges spatial reasoning and object manipulation.arXiv preprint arXiv:2502.13143, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.727266Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:08a1473422ce2715d1a0cc33f5cffa424a54572cfcbdbe9bb8377afa22750da9","observation_id":"3218ec8b-1e33-4470-9fbc-28ae5e08c84a","resolution":{"observed_at":"2026-08-07T00:57:29.727266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09783","last_updated":"2025-01-16T18:59:51Z","snapshot_observed_at":"2026-07-06T20:22:09.358394Z","submitted_at":"2025-01-16T18:59:51Z","title":"GeoManip: Geometric Constraints as General Interfaces for Robot Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09783","snapshot_observed_at":"2026-08-07T00:57:29.731232Z","title":"Geomanip: Geometric constraints as general interfaces for robot manipulation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.731232Z"},"links":{"cited_paper":"/paper/2501.09783","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:f1b7ec2542a98137d39e4ac93fca34de81aa6d069b8870cef3099d228ce12abe","observation_id":"0dc6f32f-0d3d-46c8-a402-d3623c0da8b5","resolution":{"observed_at":"2026-08-07T00:57:29.731232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10546","last_updated":"2025-03-13T16:59:17Z","snapshot_observed_at":"2026-08-02T02:57:59.885850Z","submitted_at":"2025-03-13T16:59:17Z","title":"KUDA: Keypoints to Unify Dynamics Learning and Visual Prompting for Open-Vocabulary Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10546","snapshot_observed_at":"2026-08-07T00:57:29.734890Z","title":"Kuda: Keypoints to unify dynamics learning and visual prompting for open-vocabulary robotic manipulation.arXiv preprint arXiv:2503.10546, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.734890Z"},"links":{"cited_paper":"/paper/2503.10546","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:1a0223878b9fde3e27d2df7801b10eca4567ff50092ba0dcf4db92981a7cb66e","observation_id":"3423f687-670e-46b2-9fb0-066bba60553d","resolution":{"observed_at":"2026-08-07T00:57:29.734890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11839","last_updated":"2025-08-03T09:06:58Z","snapshot_observed_at":"2026-08-04T14:15:13.344742Z","submitted_at":"2024-11-18T18:58:03Z","title":"RoboGSim: A Real2Sim2Real Robotic Gaussian Splatting Simulator","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11839","snapshot_observed_at":"2026-08-07T00:57:29.738813Z","title":"Robogsim: A real2sim2real robotic gaussian splatting simulator","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.738813Z"},"links":{"cited_paper":"/paper/2411.11839","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:cb1e51e941f9d9b57c36c8c80453c849ca0dada2b93de708bae2c97f11c666a0","observation_id":"2e2bcb92-9f27-4b04-b0d1-e5669a48fce2","resolution":{"observed_at":"2026-08-07T00:57:29.738813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.20291","last_updated":"2025-02-22T13:27:04Z","snapshot_observed_at":"2026-07-06T19:24:35.007698Z","submitted_at":"2024-09-30T13:52:05Z","title":"RL-GSBridge: 3D Gaussian Splatting Based Real2Sim2Real Method for Robotic Manipulation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.20291","snapshot_observed_at":"2026-08-07T00:57:29.743022Z","title":"Rl-gsbridge: 3d gaussian splatting based real2sim2real method for robotic manipulation learning.arXiv preprint arXiv:2409.20291, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.743022Z"},"links":{"cited_paper":"/paper/2409.20291","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:ec43cc885ee0e3a474ebb94e55e602e0e10c36aef0034ab522828734d162397b","observation_id":"da8c1c21-aa6f-476c-9a6d-610b252029e3","resolution":{"observed_at":"2026-08-07T00:57:29.743022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15937","last_updated":"2025-02-21T21:04:47Z","snapshot_observed_at":"2026-08-04T02:34:33.871046Z","submitted_at":"2025-02-21T21:04:47Z","title":"Discovery and Deployment of Emergent Robot Swarm Behaviors via Representation Learning and Real2Sim2Real Transfer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15937","snapshot_observed_at":"2026-08-07T00:57:29.746981Z","title":"Discovery and deployment of emergent robot swarm behaviors via represen- tation learning and real2sim2real transfer.arXiv preprint arXiv:2502.15937, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.746981Z"},"links":{"cited_paper":"/paper/2502.15937","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:843a16493d81cffb4f2facbe59f80b9f28386f89560984987ec9c3d306d4c431","observation_id":"f3ef6282-fc0d-4948-b435-6dd960cd8532","resolution":{"observed_at":"2026-08-07T00:57:29.746981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03949","last_updated":"2024-11-24T02:02:33Z","snapshot_observed_at":"2026-07-06T17:40:37.837709Z","submitted_at":"2024-03-06T18:55:36Z","title":"Reconciling Reality through Simulation: A Real-to-Sim-to-Real Approach for Robust Manipulation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03949","snapshot_observed_at":"2026-08-07T00:57:29.750649Z","title":"Reconciling reality through simulation: A real-to-sim-to-real approach for robust manipulation.arXiv preprint arXiv:2403.03949, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.750649Z"},"links":{"cited_paper":"/paper/2403.03949","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:f75162bcaf7a899f65a593c7fd735f111423be2c490e8329580f688991f89d67","observation_id":"ff4c716f-cc6d-4892-9889-7aff6367eb9d","resolution":{"observed_at":"2026-08-07T00:57:29.750649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.754394Z","title":"Rl-vigen: A reinforcement learning benchmark for visual generalization.Advances in Neural Information Processing Systems, 36:6720–6747, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.754394Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:a571bb27b9428b0bde52e290658ff00f78ed3da94db60ddf500f785c2b7d44dc","observation_id":"202612d6-a7d4-44a6-bd43-0bc2841f8da3","resolution":{"observed_at":"2026-08-07T00:57:29.754394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.10470","last_updated":"2021-08-25T23:42:59Z","snapshot_observed_at":"2026-07-06T11:40:56.544714Z","submitted_at":"2021-08-24T01:38:11Z","title":"Isaac Gym: High Performance GPU-Based Physics Simulation For Robot Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.10470","snapshot_observed_at":"2026-08-07T00:57:29.757961Z","title":"Isaac gym: High performance gpu-based physics simulation for robot learning.arXiv preprint arXiv:2108.10470, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.757961Z"},"links":{"cited_paper":"/paper/2108.10470","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:46e45bf190a61415dc6f279418a6fa58e9e512aebde65c1ce8a0451ee9e3c995","observation_id":"81530520-73f9-46c8-91f4-6b365c713e3a","resolution":{"observed_at":"2026-08-07T00:57:29.757961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.761706Z","title":"Efficient real2sim2real of continuum robots using deep reinforcement learning with koopman operator.IEEE Transactions on Industrial Electronics, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.761706Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:b28344bf61ecb19a06a59b857ce3d0fadfdcbd1555886cff58ed35b397621ea2","observation_id":"496335a6-57fe-43fc-b268-e855daf8b95b","resolution":{"observed_at":"2026-08-07T00:57:29.761706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.765399Z","title":"Real-time per- ception meets reactive motion generation.IEEE Robotics and Automation Letters, 3(3):1864– 1871, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.765399Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:fee6f3a23736f08ef6130a394f177b34afcc42bcbdcba8b6d78e68f04d70fef8","observation_id":"ffbbd005-5e6d-4c1a-a83c-00dca03b16e1","resolution":{"observed_at":"2026-08-07T00:57:29.765399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.12716","last_updated":"2022-05-06T14:46:46Z","snapshot_observed_at":"2026-08-07T11:18:45.003229Z","submitted_at":"2022-01-30T03:59:14Z","title":"You Only Demonstrate Once: Category-Level Manipulation from Single Visual Demonstration","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.12716","snapshot_observed_at":"2026-08-07T00:57:29.768816Z","title":"You only demonstrate once: Category-level manipulation from single visual demonstration.arXiv preprint arXiv:2201.12716, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.768816Z"},"links":{"cited_paper":"/paper/2201.12716","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:a32e0237e6cc6342287e077f4d5851a6d15a53580cbb5c1361e1783cd26970e2","observation_id":"f3f35635-23cf-4e22-9e02-b007767edf48","resolution":{"observed_at":"2026-08-07T00:57:29.768816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.773481Z","title":"One-2-3-45++: Fast single image to 3d objects with consistent multi-view generation and 3d diffusion","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.773481Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:a24d47dfffe9b7496c00383242c5850d0389a56d2a7eaf0ae512db08f6e1839c","observation_id":"f574453c-c930-4e62-9c01-d6bda8e08740","resolution":{"observed_at":"2026-08-07T00:57:29.773481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.777217Z","title":"Sparp: Fast 3d object reconstruction and pose estimation from sparse views","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.777217Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:8acb51cc9930cb15eccb78e4abdcc8618b89f1fd3eccdd01cc4f6a05118d196f","observation_id":"adbc6daf-938c-4315-bfd0-5d33f89eb69c","resolution":{"observed_at":"2026-08-07T00:57:29.777217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15110","last_updated":"2023-10-23T17:18:59Z","snapshot_observed_at":"2026-07-06T16:37:19.963994Z","submitted_at":"2023-10-23T17:18:59Z","title":"Zero123++: a Single Image to Consistent Multi-view Diffusion Base Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.15110","snapshot_observed_at":"2026-08-07T00:57:29.780816Z","title":"Zero123++: a single image to consistent multi-view diffusion base model.arXiv preprint arXiv:2310.15110, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.780816Z"},"links":{"cited_paper":"/paper/2310.15110","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:92180a37278d7c7b91a3626675e9ca360cdb7eecb64646eed8b6f1c395245a66","observation_id":"a35b8401-4e81-4ef2-bf10-f1589bf1bded","resolution":{"observed_at":"2026-08-07T00:57:29.780816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.784567Z","title":"Zero-1-to-3: Zero-shot one image to 3d object","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.784567Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:abe7719401cf1c6560ddec1fb8ff6fd40487aa61f4ae0b4e238469f2e1e67180","observation_id":"dc0a08ef-866b-441d-8562-1e5ae190eb2a","resolution":{"observed_at":"2026-08-07T00:57:29.784567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.788055Z","title":"Get3d: A generative model of high quality 3d textured shapes learned from images.Advances In Neural Information Processing Systems, 35:31841–31854, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.788055Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:ca7f8cc45e0efc2e88157e4a69e9855722ab2f52f585fc89ee5811484e193695","observation_id":"ef4dcf57-55e1-491d-864f-80c272ffa6e7","resolution":{"observed_at":"2026-08-07T00:57:29.788055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.791944Z","title":"A-sdf: Learning disentangled signed distance functions for articulated shape represen- tation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.791944Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:08191134778b6ba0fd2b13f996a970e6037ec8020fc5a2f02c14070cf0aadd01","observation_id":"48bc8d56-a630-4be9-957c-79ecffa15e31","resolution":{"observed_at":"2026-08-07T00:57:29.791944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.796122Z","title":"Ditto: Building digital twins of articulated objects from interaction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.796122Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:4b30e774e89b32f29b38708f585a9de6951b53273723082d8102eb8032586a1d","observation_id":"7580dbdc-6cdc-4ed1-9099-4bee1a27c0e8","resolution":{"observed_at":"2026-08-07T00:57:29.796122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.08997","last_updated":"2023-04-07T16:49:33Z","snapshot_observed_at":"2026-08-02T08:23:58.100005Z","submitted_at":"2022-07-19T00:27:36Z","title":"Structure from Action: Learning Interactions for Articulated Object 3D Structure Discovery","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.08997","snapshot_observed_at":"2026-08-07T00:57:29.799748Z","title":"Structure from action: Learn- ing interactions for articulated object 3d structure discovery.arXiv preprint arXiv:2207.08997, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.799748Z"},"links":{"cited_paper":"/paper/2207.08997","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:d85939618117d4f66f34f1a0c4e61c299af72b84c4bfdda17f7141c52ae3297e","observation_id":"fa0071f9-0ff6-4537-83a7-1eb67104ae13","resolution":{"observed_at":"2026-08-07T00:57:29.799748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11656","last_updated":"2024-05-31T16:44:06Z","snapshot_observed_at":"2026-07-06T18:16:29.163229Z","submitted_at":"2024-05-19T20:01:29Z","title":"URDFormer: A Pipeline for Constructing Articulated Simulation Environments from Real-World Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11656","snapshot_observed_at":"2026-08-07T00:57:29.803567Z","title":"Urdformer: A pipeline for constructing articulated simulation environments from real-world images.arXiv preprint arXiv:2405.11656, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.803567Z"},"links":{"cited_paper":"/paper/2405.11656","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:fa06814e91390fa87158cf59dbcce2fcff43d4633484c0f8a3af7181a09cbcfb","observation_id":"c419f85c-0346-4210-87fb-e8a8ed36e284","resolution":{"observed_at":"2026-08-07T00:57:29.803567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.807593Z","title":"Paris: Part-level reconstruction and motion analysis for articulated objects","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.807593Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:0ef588d0488246edd9af801499e7da671d73cce735dc46795963a76988024728","observation_id":"20fce2ed-0716-448f-a67f-d57b6c645845","resolution":{"observed_at":"2026-08-07T00:57:29.807593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.811629Z","title":"Cage: controllable articulation generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.811629Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:ad7ae6a092e9a4af6ee12e61f6d900ac7043e5f57a80d6b3fa0697646e218ef6","observation_id":"495fe785-5f7c-426a-92ef-5734e6c41c7b","resolution":{"observed_at":"2026-08-07T00:57:29.811629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.816117Z","title":"Sam-6d: Segment anything model meets zero-shot 6d object pose estimation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.816117Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:6d2de946eb439cbb3f6575f45350a667175628ea880b4a76afb45a34e8d161fe","observation_id":"0fcd9752-97bc-425d-ad36-2a43b6db519d","resolution":{"observed_at":"2026-08-07T00:57:29.816117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.819976Z","title":"Gigapose: Fast and robust novel object pose estimation via one correspondence","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.819976Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:39a766155350e0ddcad785fdc2f77871f9b1901030682b1fe401af0fe7d940ea","observation_id":"40a5fc99-2b26-40a1-876d-f1d74922802f","resolution":{"observed_at":"2026-08-07T00:57:29.819976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18673","last_updated":"2025-03-25T06:18:47Z","snapshot_observed_at":"2026-08-04T06:44:25.911218Z","submitted_at":"2025-03-24T13:46:21Z","title":"Any6D: Model-free 6D Pose Estimation of Novel Objects","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18673","snapshot_observed_at":"2026-08-07T00:57:29.823718Z","title":"Any6d: Model-free 6d pose estimation of novel objects.arXiv preprint arXiv:2503.18673, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.823718Z"},"links":{"cited_paper":"/paper/2503.18673","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:fa730a91fa33fbdad6306e3729db791d6c7b6154de8695dcc3d5a07de88e2be6","observation_id":"c38baffb-291d-483c-bcbb-d7590ebed77e","resolution":{"observed_at":"2026-08-07T00:57:29.823718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.827851Z","title":"Foundationpose: Unified 6d pose estimation and tracking of novel objects","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.827851Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:fc1c6d99b40425a35cf65d12c21b23f061e8f59b21c9778e03ceb19672b25d4b","observation_id":"3d8d1f3f-b4e2-4abb-9faa-061c4988e054","resolution":{"observed_at":"2026-08-07T00:57:29.827851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.831798Z","title":"Foundpose: Unseen object pose estimation with foundation features","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.831798Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:e91fbbc0b013832c84093e4495c596b89e0c0c4cc3545fe3800e382ff7abe6d7","observation_id":"ef06b616-a511-492c-a899-22c81d92154f","resolution":{"observed_at":"2026-08-07T00:57:29.831798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.02690","last_updated":"2021-03-05T05:59:07Z","snapshot_observed_at":"2026-08-04T03:43:34.316114Z","submitted_at":"2021-03-03T21:17:06Z","title":"A comprehensive survey on point cloud registration","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.02690","snapshot_observed_at":"2026-08-07T00:57:29.836289Z","title":"A comprehensive survey on point cloud registration.arXiv preprint arXiv:2103.02690, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.836289Z"},"links":{"cited_paper":"/paper/2103.02690","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:5c6598e7a60991636dba28b0099e94ac1c1bbc30d7407abb58e667de517a8427","observation_id":"175bfc2c-8433-43e3-80fc-0818e56a3ebf","resolution":{"observed_at":"2026-08-07T00:57:29.836289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13830","last_updated":"2025-02-02T04:32:17Z","snapshot_observed_at":"2026-07-06T18:03:29.074967Z","submitted_at":"2024-04-22T02:05:15Z","title":"Deep Learning-Based Point Cloud Registration: A Comprehensive Survey and Taxonomy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13830","snapshot_observed_at":"2026-08-07T00:57:29.840574Z","title":"Deep learning-based point cloud registration: A comprehensive survey and taxonomy","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.840574Z"},"links":{"cited_paper":"/paper/2404.13830","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:0818ae806f08d31a12868a4278e00b530c0f7c564df538233e91cbd37e0e8c09","observation_id":"64a7666d-48b9-4c6c-a827-03d561bd4fd3","resolution":{"observed_at":"2026-08-07T00:57:29.840574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.844471Z","title":"A tutorial review on point cloud registrations: principle, classification, comparison, and technology challenges.Mathematical Problems in Engineering, 2021(1):9953910, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.844471Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:9df84ba07d9e739999bda11d88bb56ca5f67db3fecb079cec01bb6d348cb5bcd","observation_id":"4b60822b-1233-45b1-a150-d639ee99a26c","resolution":{"observed_at":"2026-08-07T00:57:29.844471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.848687Z","title":"A comprehensive survey of visual slam algorithms.Robotics, 11(1):24, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.848687Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:96929e1b024fbbd8291beba92d780623f4869df011e9e81bb5cc7cea45790098","observation_id":"6ad01dc7-36e0-4509-a66b-54ff429948b6","resolution":{"observed_at":"2026-08-07T00:57:29.848687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13255","last_updated":"2025-03-27T14:03:25Z","snapshot_observed_at":"2026-07-06T17:33:00.716054Z","submitted_at":"2024-02-20T18:59:57Z","title":"How NeRFs and 3D Gaussian Splatting are Reshaping SLAM: a Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13255","snapshot_observed_at":"2026-08-07T00:57:29.852481Z","title":"How nerfs and 3d gaussian splatting are reshaping slam: a survey.arXiv preprint arXiv:2402.13255, 4:1, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.852481Z"},"links":{"cited_paper":"/paper/2402.13255","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:72cf4f108a3def52149be4d4d020c23bb4eea36b64a735435e6281c63b976b51","observation_id":"64a42648-e64e-4826-9f97-f59431350a19","resolution":{"observed_at":"2026-08-07T00:57:29.852481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.856398Z","title":"A survey on active simultaneous localization and mapping: State of the art and new frontiers.IEEE Transactions on Robotics, 39(3):1686–1705, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.856398Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:b30bb6350817bd0e50ca1ed74156d7f456b8b4a5d4f7d9b8a7a0ed5a372d9708","observation_id":"15c6e1a5-f68c-4fa6-832e-18218550afdc","resolution":{"observed_at":"2026-08-07T00:57:29.856398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.00370","last_updated":"2025-04-01T03:01:18Z","snapshot_observed_at":"2026-07-06T20:44:49.893232Z","submitted_at":"2025-03-01T06:40:41Z","title":"Scalable Real2Sim: Physics-Aware Asset Generation Via Robotic Pick-and-Place Setups","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.00370","snapshot_observed_at":"2026-08-07T00:57:29.859946Z","title":"Scalable real2sim: Physics-aware asset generation via robotic pick-and-place setups.arXiv preprint arXiv:2503.00370, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.859946Z"},"links":{"cited_paper":"/paper/2503.00370","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:87755e224fc34e4fe86333930694a262b34e31c80c09c121f2efaed0d904059d","observation_id":"533d08e4-5d2e-4e78-ad27-85348be21404","resolution":{"observed_at":"2026-08-07T00:57:29.859946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17973","last_updated":"2025-03-23T07:49:19Z","snapshot_observed_at":"2026-07-06T20:57:16.750503Z","submitted_at":"2025-03-23T07:49:19Z","title":"PhysTwin: Physics-Informed Reconstruction and Simulation of Deformable Objects from Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.17973","snapshot_observed_at":"2026-08-07T00:57:29.863672Z","title":"Phystwin: Physics-informed reconstruction and simulation of deformable objects from videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.863672Z"},"links":{"cited_paper":"/paper/2503.17973","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:50f52c667bfdba7689048a86427b95147d9337c6260721e954a807f028898c15","observation_id":"5b6072d7-adf1-4bd5-a78c-54b04a71c4d0","resolution":{"observed_at":"2026-08-07T00:57:29.863672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.867354Z","title":"Sim2real 2: Actively building explicit physics model for precise articulated object manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.867354Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:7aa66c6c8045409e17ec56a6979845bd7421513f3a27267d7056da1402a6a60d","observation_id":"daec14fd-34fd-4737-8cad-94abc14cff19","resolution":{"observed_at":"2026-08-07T00:57:29.867354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.870998Z","title":"A real2sim2real method for robust object grasping with neural surface reconstruction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.870998Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:076a1a42e402737ed2e9682810679e293c41e9e9099362a4edd982b94ed7fc80","observation_id":"de7bba1a-3309-4736-8a1a-7dda6b800c41","resolution":{"observed_at":"2026-08-07T00:57:29.870998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04689","last_updated":"2025-01-08T18:52:03Z","snapshot_observed_at":"2026-08-07T12:36:45.592859Z","submitted_at":"2025-01-08T18:52:03Z","title":"SPAR3D: Stable Point-Aware Reconstruction of 3D Objects from Single Images","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04689","snapshot_observed_at":"2026-08-07T00:57:29.874470Z","title":"Spar3d: Stable point-aware reconstruction of 3d objects from single images.arXiv preprint arXiv:2501.04689, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.874470Z"},"links":{"cited_paper":"/paper/2501.04689","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:2ab9ed9287ec1612e5f680f210b65676a529e4b057257314f4c27b6bd2a3c7cc","observation_id":"b42e82ff-3951-4d4f-ae39-dffb6af3f622","resolution":{"observed_at":"2026-08-07T00:57:29.874470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.878448Z","title":"Pointllm: Empowering large language models to understand point clouds","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.878448Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:90f5fb8a3f802b84e0b0d1ef9dcd28a38065cdfe760a567d7969bc9b35a833f7","observation_id":"f01ede5d-9fa8-446c-bdc7-6d8f221cebb1","resolution":{"observed_at":"2026-08-07T00:57:29.878448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.882105Z","title":"Point-nerf: Point-based neural radiance fields","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.882105Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:77d4235f286bb2eb5b5f56aa2946401277a2053007097c925692f76457d84bcc","observation_id":"a181ef1d-f192-48db-ae2a-fa10fe1ae6ef","resolution":{"observed_at":"2026-08-07T00:57:29.882105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.885459Z","title":"Pointclip: Point cloud understanding by clip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.885459Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:8917031156730e17a1bbf0b88196562da69ee1f5576dd1928f7fa09fbb0c5f15","observation_id":"df7441fc-37c6-4163-94d6-9fcbeddb9643","resolution":{"observed_at":"2026-08-07T00:57:29.885459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.889110Z","title":"Text2nerf: Text-driven 3d scene generation with neural radiance fields.IEEE Transactions on Visualization and Computer Graphics, 30(12):7749–7762, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.889110Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:79fad051712e63894c98099f2b3564c9d040bca85bbc2a2d5a6fdf0c38c08fe5","observation_id":"61384a05-d953-4a85-8b61-2d7e14236333","resolution":{"observed_at":"2026-08-07T00:57:29.889110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.893362Z","title":"Pointr: Diverse point cloud completion with geometry-aware transformers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.893362Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:c4de5d74e711a0a5607817284880ead7a6f51285d06674d36616a666933feea7","observation_id":"6086479a-47b4-4b7c-9399-15e7e233b747","resolution":{"observed_at":"2026-08-07T00:57:29.893362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:57:29.897079Z","title":"V oxel set transformer: A set-to-set approach to 3d object detection from point clouds","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:29.897079Z"},"links":{"citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:1145281a9e58a89dfdc21510b1ed33474d5b18470e00792072c0690d1de56b2f","observation_id":"613b9ae0-f12c-4977-9baa-8f78b495a719","resolution":{"observed_at":"2026-08-07T00:57:29.897079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","latest_version":2,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-07T07:38:26.007252Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":99,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":137},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 137 outbound references and 0 inbound Pith citation observations for arXiv:2506.12374."}