{"as_of":"2026-08-21T05:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:799a6ab79ae794f89facb0c478d45a15016ea034d32b70b9bb47a637dc486038","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T00:36:56.172379Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.11739/citation-record","integrity":"/paper/2608.11739/integrity","json":"/paper/2608.11739/citation-record.json","paper":"/paper/2608.11739"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:55.912331Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.912331Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:3ce808097088dba91568e4688d1a9613580e03358051868021ec0bc896da1976","observation_id":"e6a4538b-0d66-4493-b259-442553d7553a","resolution":{"observed_at":"2026-08-16T00:36:55.912331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-08-16T21:53:14.144225Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-16T00:36:55.917278Z","title":"Openvla: An open-source vision-language- action model.arXiv preprint arXiv:2406.09246, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.917278Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:5f986fa6816a94fbd900dc6aa70ce5a682763de08cc2bf52433c6908fda94f52","observation_id":"ca02fd4b-c510-4933-be4d-b64467a05af3","resolution":{"observed_at":"2026-08-16T00:36:55.917278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-08-16T17:53:54.636855Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-16T00:36:55.922396Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.922396Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:02e3e7cd87dcd37569ec8865015b13090ec2e83b8af16fa3ecdf8fde3c1c78c0","observation_id":"51933ffc-6a2d-4d64-b315-b1ec127ea0d9","resolution":{"observed_at":"2026-08-16T00:36:55.922396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16054","last_updated":"2025-04-22T17:31:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:31:29Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16054","snapshot_observed_at":"2026-08-16T00:36:55.926962Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.926962Z"},"links":{"cited_paper":"/paper/2504.16054","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:1409ecab4f06b110dd1b80c3e195ed60c9c3e3fd9d1683e8ede754722b209643","observation_id":"02353279-4265-4c9e-b1e8-d512588509bd","resolution":{"observed_at":"2026-08-16T00:36:55.926962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14734","last_updated":"2025-03-27T02:52:43Z","snapshot_observed_at":"2026-08-02T04:15:31.100670Z","submitted_at":"2025-03-18T21:06:21Z","title":"GR00T N1: An Open Foundation Model for Generalist Humanoid Robots","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14734","snapshot_observed_at":"2026-08-16T00:36:55.931882Z","title":"Gr00t n1: An open foundation model for generalist humanoid robots.arXiv preprint arXiv:2503.14734, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.931882Z"},"links":{"cited_paper":"/paper/2503.14734","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:ed87ce545beadfe03885f72b39e1216a626b74eda9f1a66910623a4f983d1e50","observation_id":"d23df188-45d1-43f4-a669-643c936e91e4","resolution":{"observed_at":"2026-08-16T00:36:55.931882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01844","last_updated":"2025-06-02T16:30:19Z","snapshot_observed_at":"2026-08-18T04:25:30.115862Z","submitted_at":"2025-06-02T16:30:19Z","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01844","snapshot_observed_at":"2026-08-16T00:36:55.936631Z","title":"Smolvla: A vision-language- action model for affordable and efficient robotics.arXiv preprint arXiv:2506.01844, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.936631Z"},"links":{"cited_paper":"/paper/2506.01844","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:4b9cff368cc65d689c5a3672153f0f1019d553ec20fa7f22178f5b233f242c51","observation_id":"7ee44f63-a2a0-42c6-bc90-a411cc0b2f02","resolution":{"observed_at":"2026-08-16T00:36:55.936631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.744730Z","title":"Cot-vla: Visualchain-of-thoughtreasoningforvision-language-action models","venue":null,"work_id":"247e505c-5500-437d-a099-0ec8a92bcfd4","year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.941670Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:832c81c859e0e9d12066a77662d4e2080d4b410ca8662b923451f89caf5dd35a","observation_id":"6e441feb-240b-4ce5-bf4e-1c0d1232238b","resolution":{"observed_at":"2026-08-16T00:36:57.749609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:55.946552Z","title":"Dualcot-vla: Visual-linguistic chain of thought via parallel reasoning for vision-language-action models.arXiv preprint arXiv:2603.22280, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.946552Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:9f2c0860dc17e8cca6378ca025198f6def0cbc4ed5449fbc34514c0495353d47","observation_id":"81937156-f8c8-4258-9fd5-d8b916e73a54","resolution":{"observed_at":"2026-08-16T00:36:55.946552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:55.950855Z","title":"Halo: A unified vision-language-action model for embodied multimodal chain-of-thought reasoning.arXiv preprint arXiv:2602.21157, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.950855Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:494b45efe485f91c97a55095bc8db090827afff7d181000c1c2baa1d2a75e5b2","observation_id":"0df9b212-c39b-443a-99c4-191cd45f70da","resolution":{"observed_at":"2026-08-16T00:36:55.950855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:55.955170Z","title":"Mem: Multi-scale embodied memory for vision language action models.arXiv preprint arXiv:2603.03596, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.955170Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:11e4b9f5ffb294559750d650b1f7ae990660e2d36f66e0ade9682d6896e21efc","observation_id":"71786589-ee15-4fe5-82c8-6a46adf63cc0","resolution":{"observed_at":"2026-08-16T00:36:55.955170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09747","last_updated":"2025-01-16T18:57:04Z","snapshot_observed_at":"2026-08-21T05:15:36.931588Z","submitted_at":"2025-01-16T18:57:04Z","title":"FAST: Efficient Action Tokenization for Vision-Language-Action Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09747","snapshot_observed_at":"2026-08-16T00:36:55.959246Z","title":"Fast: Efficient action tokenization for vision-language-action models.arXiv preprint arXiv:2501.09747, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.959246Z"},"links":{"cited_paper":"/paper/2501.09747","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:ff34edafd69b54c1c85da33df0136594ade206c7ce932192af17977f8ccef3fb","observation_id":"d176f7ec-a79f-454b-9802-63139c1eb7ad","resolution":{"observed_at":"2026-08-16T00:36:55.959246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:55.963973Z","title":"Flowvla: Visual chain of thought-based motion reasoning for vision-language-action models.arXiv preprint arXiv:2508.18269, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.963973Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:29a122765b63b80dcde26ba5d3d21bee1c6479498c44762bd242f01bfcb1ff45","observation_id":"7c8e958b-89b0-421e-bc2c-72cdb841c485","resolution":{"observed_at":"2026-08-16T00:36:55.963973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.729936Z","title":"BEHAVIOR-1K: A benchmark for embodied ai with 1,000 everyday activities and realistic simulation","venue":null,"work_id":"ccb16428-0988-4299-9f4e-d870b6ad0dde","year":2023},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.968316Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:1d625adce864a042a4fbfaaf59521fed3398a8ff45556469c488f69234bbd066","observation_id":"c40c6bab-be37-45e5-b986-223bac6ea4aa","resolution":{"observed_at":"2026-08-16T00:36:57.734827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.715572Z","title":"LIBERO: Benchmarking knowledge transfer for lifelong robot learning.Advances in Neural Information Processing Systems, 36:44776–44791, 2023","venue":null,"work_id":"8cfd6123-29fa-4e97-b84f-c054739c581f","year":2023},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.972164Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:4892e4609f697bfabf500b024b3287ebf42386667e99bd43d1d294dfa0799655","observation_id":"e1d83742-33f5-4f8d-ae2a-aad23cb452f5","resolution":{"observed_at":"2026-08-16T00:36:57.720481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.05941","last_updated":"2024-05-09T17:30:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-09T17:30:16Z","title":"Evaluating Real-World Robot Manipulation Policies in Simulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.05941","snapshot_observed_at":"2026-08-16T00:36:55.976306Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.976306Z"},"links":{"cited_paper":"/paper/2405.05941","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:3d64b1a120b62fb48f03f5aaab136c46861a6d8e5fead32d461d189d89a2c2b2","observation_id":"7b33b106-c09e-4051-8f90-173be0b8b1c0","resolution":{"observed_at":"2026-08-16T00:36:55.976306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.16666","last_updated":"2026-03-23T05:41:14Z","snapshot_observed_at":"2026-08-20T03:53:25.888886Z","submitted_at":"2026-03-17T15:33:43Z","title":"Fast-WAM: Do World Action Models Need Test-time Future Imagination?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.16666","snapshot_observed_at":"2026-08-16T00:36:55.980689Z","title":"Fast-wam: Do world action models need test-time future imagination?arXiv preprint arXiv:2603.16666, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.980689Z"},"links":{"cited_paper":"/paper/2603.16666","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:efb9da72604f81678bb1dd608f0e73d3daa48b65382598cf484dec6f98fd61d5","observation_id":"dc35eb31-beca-464b-93dc-5e949eb3f3b5","resolution":{"observed_at":"2026-08-16T00:36:55.980689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.701098Z","title":"Knowledge insulating vision-language-action models: Trainfast, runfast, generalizebetter.AdvancesinNeuralInformationProcessingSystems, 38:102867–102888, 2026","venue":null,"work_id":"f3a51821-b7fe-4635-8a82-8f90d0cb86f5","year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.984853Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:48dfbe7145bafeef89a0256ecc8e1fcdd9b43d8b3b720d1db07f25690731288b","observation_id":"5ab3b970-d16d-4385-9d8c-40280cfee4cf","resolution":{"observed_at":"2026-08-16T00:36:57.705781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:55.989259Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.989259Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:44c5495d143781b53b92d535e4b7b2ea254623d1ab28e2e5354c6caac54b8df4","observation_id":"118d3919-840b-41ad-bf6d-a962e1b4a4c7","resolution":{"observed_at":"2026-08-16T00:36:55.989259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:55.993408Z","title":"Vq-vla: Improving vision-language-action models via scaling vector-quantized action tokenizers","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.993408Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:4b1145027786e4eb69ffa1fb22a0cca018c7cd89f318652f5392e8b1c6e830a9","observation_id":"7ff129e5-8f3a-414f-ae96-3b0998c99f15","resolution":{"observed_at":"2026-08-16T00:36:55.993408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.676886Z","title":"Beast: Efficient tokenization of b-splines encoded action sequences for imitation learning.Advances in Neural Information Processing Systems, 38:172934–172959, 2026","venue":null,"work_id":"619f18f8-7e17-4252-a6c9-f33751b83330","year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:55.997542Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:66d76ce9f4120b1fcdc6d4e9824c2ad4236dcc608d675301d19fc636d16a0379","observation_id":"ea4e75db-a95c-4866-8bb0-420c4973051e","resolution":{"observed_at":"2026-08-16T00:36:57.681457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03181","last_updated":"2024-06-28T04:15:33Z","snapshot_observed_at":"2026-08-16T14:12:33.322650Z","submitted_at":"2024-03-05T18:19:29Z","title":"Behavior Generation with Latent Actions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03181","snapshot_observed_at":"2026-08-16T00:36:56.001612Z","title":"Behavior generation with latent actions.arXiv preprint arXiv:2403.03181, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.001612Z"},"links":{"cited_paper":"/paper/2403.03181","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:9db897f8dcfe7696063e163f726d1a17c05ce6c886da1889f99431e4ee9e8d45","observation_id":"e2732b29-346a-459a-899c-54589328e9bb","resolution":{"observed_at":"2026-08-16T00:36:56.001612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15830","last_updated":"2025-05-19T02:40:18Z","snapshot_observed_at":"2026-08-20T05:57:34.826209Z","submitted_at":"2025-01-27T07:34:33Z","title":"SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15830","snapshot_observed_at":"2026-08-16T00:36:56.006016Z","title":"Spatialvla: Exploring spatial representations for visual-language-action model","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.006016Z"},"links":{"cited_paper":"/paper/2501.15830","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:73857e81cd671f26293cc16a4f6cd7afe83335dac68137d73f094aeb2bfebaec","observation_id":"3b081125-96cd-41cd-b48d-7c7014fc943a","resolution":{"observed_at":"2026-08-16T00:36:56.006016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.010632Z","title":"Being-h0","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.010632Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:02c1f2de632e78fa643ca70eb52fce60a098bf764e145595d672097b36548805","observation_id":"4770f8a3-0545-4b1f-9cef-9f5944c71d0d","resolution":{"observed_at":"2026-08-16T00:36:56.010632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.014811Z","title":"Green-vla: Staged vision-language-action model for generalist robots","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.014811Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:a79b1bfc978470d258de7cc221229b5a3b9e7b51b353e2a4307def962f80ce21","observation_id":"b97e535f-9489-41da-a876-6ee918418af3","resolution":{"observed_at":"2026-08-16T00:36:56.014811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.07993","last_updated":"2026-05-19T11:30:49Z","snapshot_observed_at":"2026-08-11T09:58:02.434831Z","submitted_at":"2026-04-09T09:01:43Z","title":"HEX: Humanoid-Aligned Experts for Cross-Embodiment Whole-Body Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.07993","snapshot_observed_at":"2026-08-16T00:36:56.019095Z","title":"Hex: Humanoid-aligned experts for cross-embodiment whole-body manipulation.arXiv preprint arXiv:2604.07993, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.019095Z"},"links":{"cited_paper":"/paper/2604.07993","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:5e9eebce739059ba8b282b1d63683a136eab3aa679d3ca54d5513afaff843fca","observation_id":"21573125-87f0-453f-80c9-339104c92188","resolution":{"observed_at":"2026-08-16T00:36:56.019095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.662299Z","title":"Hamster: Hierarchicalactionmodelsforopen-worldrobotmanipulation","venue":null,"work_id":"ca39c2db-2284-412b-9f8e-9e5f70d332cb","year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.024803Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:c6244b4da2f83672a156c1f157b51bb084446f8f5de882b8eaccdeeb1844e453","observation_id":"c28e83a9-1918-4345-be6d-1c57232d5224","resolution":{"observed_at":"2026-08-16T00:36:57.667542Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08693","last_updated":"2025-03-06T19:29:03Z","snapshot_observed_at":"2026-08-07T02:44:43.738657Z","submitted_at":"2024-07-11T17:31:01Z","title":"Robotic Control via Embodied Chain-of-Thought Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08693","snapshot_observed_at":"2026-08-16T00:36:56.029284Z","title":"Robotic control via embodied chain-of-thought reasoning.arXiv preprint arXiv:2407.08693, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.029284Z"},"links":{"cited_paper":"/paper/2407.08693","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:a35a4ab3be6122bca956cd3a8595037dba8349539aac8779a164ef433587da91","observation_id":"e64ba385-8d49-4c8f-bff4-e58458c96197","resolution":{"observed_at":"2026-08-16T00:36:56.029284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.645914Z","title":"Emma-x: An embodied multimodal action model with grounded chain of thought and look-ahead spatial reasoning","venue":null,"work_id":"ce82bbb2-fcdb-4df3-bf4c-54b52577f0a4","year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.033759Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:0dc29d40f3a2fbb519a08ceb2fd85fe4e1210f6c9da4b57329270d5ff281609c","observation_id":"c1ca599c-a6eb-43b1-b760-3a5440ec15a2","resolution":{"observed_at":"2026-08-16T00:36:57.650531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.631434Z","title":"Tracevla: Visual trace prompting enhances spatial-temporal awareness for generalist robotic policies","venue":null,"work_id":"d37a671a-9946-49b9-8747-01ec5a9bad30","year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.037632Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:d9c494568b0b4c35c7a8f5f0c85d754c6dcbd055641c2b2072041cfe4289fc1a","observation_id":"dfcc9841-b045-48b1-b8ac-c5ba5ddc5af5","resolution":{"observed_at":"2026-08-16T00:36:57.636018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.617222Z","title":null,"venue":null,"work_id":"07c22de9-03c0-4605-bd80-f083c8d2e28b","year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.041973Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:bf9352cf4afc5769d14ebd3ee8fa4088eb621f461d691c288c3bcc7c67703f5f","observation_id":"0c10f5de-1a2d-4f14-ad6d-f924c74e88c5","resolution":{"observed_at":"2026-08-16T00:36:57.621745Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.046109Z","title":"Minivla: A better vla with a smaller footprint, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.046109Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:2cb372c850e165a98358ebc9d6a5ca4b2f74bcff098d65172452eca9ba1e0951","observation_id":"18d078ad-820f-431c-8f5e-11f50a2c9449","resolution":{"observed_at":"2026-08-16T00:36:56.046109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.050256Z","title":"Actioncodec: What makes for good action tokenizers.arXiv preprint arXiv:2602.15397, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.050256Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:2004e629a475543e1fad4f85a81c3b8ea017b07b17cbfbb158b70ecef8e2c742","observation_id":"55842ded-b04b-4160-9501-ab1b003a48a8","resolution":{"observed_at":"2026-08-16T00:36:56.050256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.594217Z","title":"FASTer: Toward powerful and efficient autoregressive vision–language–action models with learnableactiontokenizerandblock-wisedecoding","venue":null,"work_id":"d9b086d8-883c-412d-9abd-70e01eb30c9e","year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.054258Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:97a62db49a89f707fea79ebb3e50459e8974d6ca24dc766b556cc2030b28c63e","observation_id":"def8cd19-5178-4684-a2ad-24f0a36b3f3e","resolution":{"observed_at":"2026-08-16T00:36:57.598968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15483","last_updated":"2026-04-24T23:18:28Z","snapshot_observed_at":"2026-08-11T15:03:01.640683Z","submitted_at":"2026-04-16T19:18:07Z","title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.15483","snapshot_observed_at":"2026-08-16T00:36:56.058688Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.058688Z"},"links":{"cited_paper":"/paper/2604.15483","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:0b390e417c9471992f83fb3203f2bbfebd3171017178f9e83e8400545a67e51c","observation_id":"6330763b-1a08-4b67-98cf-93100ef78cdf","resolution":{"observed_at":"2026-08-16T00:36:56.058688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.580041Z","title":"Gemini 3 pro model card, 2026","venue":null,"work_id":"f8b6675b-318e-4b27-8c35-d03e3bbcd940","year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.062853Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:91aa89371c35474e3649a6f4f14f4b5e41b2ced24be0580c450e8a636ae9964d","observation_id":"c61b6bc6-ffad-4425-8e66-20d8d76ae290","resolution":{"observed_at":"2026-08-16T00:36:57.584095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.566203Z","title":"Seed 2.0 official launch, 2026","venue":null,"work_id":"aa36dbdd-3429-4b60-b9bc-e787a6a3d329","year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.066778Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:9f3902781b9bdd1afe288c4fe9278dcad2800c9cd2a8cd0f1c78d4e107480636","observation_id":"c5d5b08e-8d74-4bff-8c67-d665754cc486","resolution":{"observed_at":"2026-08-16T00:36:57.570693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.16719","last_updated":"2026-03-28T16:54:56Z","snapshot_observed_at":"2026-08-07T05:59:39.049027Z","submitted_at":"2025-11-20T18:59:56Z","title":"SAM 3: Segment Anything with Concepts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.16719","snapshot_observed_at":"2026-08-16T00:36:56.070976Z","title":"Sam 3: Segment anything with concepts, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.070976Z"},"links":{"cited_paper":"/paper/2511.16719","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:533577c493d161d4a58d27213f126151fec92c7b09b72b0f66823325923698b5","observation_id":"fb8ed6d8-9ab3-453f-be36-0c5c6dc4add6","resolution":{"observed_at":"2026-08-16T00:36:56.070976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.552099Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, January 2024","venue":null,"work_id":"6aea7545-7b09-44fc-bf27-1e00a14eeb39","year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.075505Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:0098d08e44e3da6c329ca5367648af1f9f6fbf813e79a7776bff8e25854fe414","observation_id":"08764ba2-cd26-4b8a-8a8e-086090018625","resolution":{"observed_at":"2026-08-16T00:36:57.556683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-08-18T11:56:50.710310Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-16T00:36:56.079605Z","title":"Llava-onevision: Easy visual task transfer, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.079605Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:7421e8ae22b5e314dbd273cad0c7d04cb707765090718f0b0b4f265b0be9db6c","observation_id":"a864ae24-7f0f-460f-9d79-cc2202f82167","resolution":{"observed_at":"2026-08-16T00:36:56.079605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10721","last_updated":"2024-06-15T19:22:51Z","snapshot_observed_at":"2026-08-16T13:42:40.054113Z","submitted_at":"2024-06-15T19:22:51Z","title":"RoboPoint: A Vision-Language Model for Spatial Affordance Prediction for Robotics","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10721","snapshot_observed_at":"2026-08-16T00:36:56.083603Z","title":"Robopoint: A vision-language model for spatial affordance prediction for robotics, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.083603Z"},"links":{"cited_paper":"/paper/2406.10721","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:45a74f0cca8b49ee3aab5ac4f3adfa2c13e19c62d15dd782127d997d6cb0c4d0","observation_id":"b12fdf99-bef0-46b6-98e9-272d5d50a9c5","resolution":{"observed_at":"2026-08-16T00:36:56.083603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.07917","last_updated":"2025-09-18T12:21:57Z","snapshot_observed_at":"2026-08-14T09:44:01.476519Z","submitted_at":"2025-08-11T12:32:45Z","title":"MolmoAct: Action Reasoning Models that can Reason in Space","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.07917","snapshot_observed_at":"2026-08-16T00:36:56.087930Z","title":"Molmoact: Action reasoning models that can reason in space, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.087930Z"},"links":{"cited_paper":"/paper/2508.07917","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:a0ed71e01b54c2cbb3b5dc6ef7658fe82f1ebf217f6c92e42745c79a5622717d","observation_id":"fcfeb2e4-5c44-42e2-a26e-8f9abcd78d03","resolution":{"observed_at":"2026-08-16T00:36:56.087930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21257","last_updated":"2025-03-25T05:46:03Z","snapshot_observed_at":"2026-08-20T16:21:51.679872Z","submitted_at":"2025-02-28T17:30:39Z","title":"RoboBrain: A Unified Brain Model for Robotic Manipulation from Abstract to Concrete","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.21257","snapshot_observed_at":"2026-08-16T00:36:56.092518Z","title":"Robobrain: A unified brain model for robotic manipulation from abstract to concrete, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.092518Z"},"links":{"cited_paper":"/paper/2502.21257","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:53b19143f986e6745feb22592b99d4ce993a42db42abf3c60c8ffc2694163a8b","observation_id":"711b78ad-9685-4331-b31d-4b7f8ba580e5","resolution":{"observed_at":"2026-08-16T00:36:56.092518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12945","last_updated":"2025-04-22T17:57:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-19T17:48:38Z","title":"DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12945","snapshot_observed_at":"2026-08-16T00:36:56.096943Z","title":"DROID: A large-scale in-the-wild robot manipulation dataset.arXiv preprint arXiv:2403.12945, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.096943Z"},"links":{"cited_paper":"/paper/2403.12945","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:09c38db7cfc4e2813a5ed2e9f4aa5622b079255b8ca3460bd85b566b2b227a67","observation_id":"20e85a7a-93e7-4229-a702-78c68aee629f","resolution":{"observed_at":"2026-08-16T00:36:56.096943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.02881","last_updated":"2026-05-08T04:21:51Z","snapshot_observed_at":"2026-08-16T22:33:00.868689Z","submitted_at":"2026-05-04T17:51:21Z","title":"MolmoAct2: Action Reasoning Models for Real-world Deployment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.02881","snapshot_observed_at":"2026-08-16T00:36:56.101107Z","title":"Molmoact2: Action reasoning models for real-world deployment, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.101107Z"},"links":{"cited_paper":"/paper/2605.02881","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:7ec836b10bc50ccd46ff7937386acb4fdffa32c614198769a5e87abb42749c02","observation_id":"cbd6ad81-2b84-43fb-998e-1a6019684428","resolution":{"observed_at":"2026-08-16T00:36:56.101107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.105340Z","title":"Bridgedata v2: A dataset for robot learning at scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.105340Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:ba7801262ff90f9b99b392b877cb9a8d2568620bb6256f0747f917e91bc74c34","observation_id":"07313b5c-7309-4835-b187-d4095547961d","resolution":{"observed_at":"2026-08-16T00:36:56.105340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.528516Z","title":"Starvla: A lego-like codebase for vision- language-action model developing, 2026","venue":null,"work_id":"f64e0a1f-16ee-40eb-bf27-a0475486ed61","year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.109557Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:b917b1a1ef52e54a54f150485ce850e6be3583fca74fe38e3334319d3eca69a0","observation_id":"f25edd1a-9048-4098-b340-bc88846a725d","resolution":{"observed_at":"2026-08-16T00:36:57.533657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:57.512031Z","title":"Memoryvla: Perceptual-cognitive memory in vision-language-action models for robotic manipulation, 2025","venue":null,"work_id":"e0b3dd98-2d00-412a-8936-1324ac1c6120","year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.113432Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:7e9abd5e939525828cd38a1e2e03f7ad462fddc6ca86d6800b86711242c759bd","observation_id":"b14255fa-8191-4c86-aebe-383f6ba8736e","resolution":{"observed_at":"2026-08-16T00:36:57.518633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.117394Z","title":"Eo-1: An open unified embodied foundation model for general robot control.arXiv preprint arXiv:2508.21112, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.117394Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:68377be490cb8c7fdc6d33d6a2b36bf6b0284bef1b4292dd518c5726353fa672","observation_id":"52e3020b-d5c2-4547-a907-2bb516040913","resolution":{"observed_at":"2026-08-16T00:36:56.117394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.121460Z","title":"Xiaomi-robotics-0: An open-sourced vision-language-action model with real-time execution.arXiv preprint arXiv:2602.12684, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.121460Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:e6d721ce575a84de10fc656b7abced726aeee8b101e715c926b762c20ff12786","observation_id":"82b131a1-4540-4122-8476-084a12e47f8d","resolution":{"observed_at":"2026-08-16T00:36:56.121460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.13030","last_updated":"2025-12-25T08:16:05Z","snapshot_observed_at":"2026-08-16T22:05:17.066613Z","submitted_at":"2025-12-15T06:58:40Z","title":"Motus: A Unified Latent Action World Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.13030","snapshot_observed_at":"2026-08-16T00:36:56.125563Z","title":"Motus: A unified latent action world model.arXiv preprint arXiv:2512.13030, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.125563Z"},"links":{"cited_paper":"/paper/2512.13030","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:9c2e0f8c49cca02bcef60b9c84dcb029ca870479cf041189f3ea43f4b8fecd4c","observation_id":"57c37a23-dd03-495c-bf76-86495884294d","resolution":{"observed_at":"2026-08-16T00:36:56.125563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.21998","last_updated":"2026-03-22T15:37:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-29T17:07:43Z","title":"Causal World Modeling for Robot Control","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.21998","snapshot_observed_at":"2026-08-16T00:36:56.129732Z","title":"Causal world modeling for robot control.arXiv preprint arXiv:2601.21998, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.129732Z"},"links":{"cited_paper":"/paper/2601.21998","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:9808621d56df71368b4c7561c3d8424a51280ed58a00f4e3e7395f6fdf97369e","observation_id":"3d70e131-f2a7-4f5c-b0b1-c165da572692","resolution":{"observed_at":"2026-08-16T00:36:56.129732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.18692","last_updated":"2026-02-26T03:30:01Z","snapshot_observed_at":"2026-08-11T08:52:29.230329Z","submitted_at":"2026-01-26T17:08:04Z","title":"A Pragmatic VLA Foundation Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.18692","snapshot_observed_at":"2026-08-16T00:36:56.133832Z","title":"A pragmatic vla foundation model.arXiv preprint arXiv:2601.18692, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.133832Z"},"links":{"cited_paper":"/paper/2601.18692","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:50d7018029d2e7ba6d38380a2ab8019c2014a3c2b47ba7100bc8ad3279cc9c1a","observation_id":"4fdca3a8-c7db-45a3-8229-e71a6c246cf1","resolution":{"observed_at":"2026-08-16T00:36:56.133832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.30280","last_updated":"2026-06-01T13:48:35Z","snapshot_observed_at":"2026-08-18T14:09:54.189121Z","submitted_at":"2026-05-28T17:36:31Z","title":"Qwen-VLA: Unifying Vision-Language-Action Modeling across Tasks, Environments, and Robot Embodiments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.30280","snapshot_observed_at":"2026-08-16T00:36:56.138129Z","title":"Qwen-vla: Unifying vision-language-action modeling across tasks, environments, and robot embodiments.arXiv preprint arXiv:2605.30280, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.138129Z"},"links":{"cited_paper":"/paper/2605.30280","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:df2667152e5cc44be7e56e9848de72417aa1c6756ccf1fb513a62f932a924c07","observation_id":"13325ed3-0e75-47b8-8ef1-5e0fbc3eaa2f","resolution":{"observed_at":"2026-08-16T00:36:56.138129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.13778","last_updated":"2025-10-15T17:30:05Z","snapshot_observed_at":"2026-07-06T22:32:47.087967Z","submitted_at":"2025-10-15T17:30:05Z","title":"InternVLA-M1: A Spatially Guided Vision-Language-Action Framework for Generalist Robot Policy","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.13778","snapshot_observed_at":"2026-08-16T00:36:56.142437Z","title":"Internvla-m1: A spatially guided vision-language-action framework for generalist robot policy.arXiv preprint arXiv:2510.13778, 2025","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.142437Z"},"links":{"cited_paper":"/paper/2510.13778","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:bae794fa53da83e514ed2f3652faf053b3684b9838f8e23742d7372cb3fd1d89","observation_id":"99d173db-18b3-4ae3-bbd6-b2c9c362a1e6","resolution":{"observed_at":"2026-08-16T00:36:56.142437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.146862Z","title":"Igniting vlms toward the embodied space.arXiv preprint arXiv:2509.11766, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.146862Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:759474514a8f0ff71ebc149648da5177af0a2230ca48397f5b765830cf5af1cc","observation_id":"7140d6cc-f9cc-433a-99f4-3cec222604c5","resolution":{"observed_at":"2026-08-16T00:36:56.146862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19645","last_updated":"2025-04-28T07:49:39Z","snapshot_observed_at":"2026-08-15T09:35:08.116329Z","submitted_at":"2025-02-27T00:30:29Z","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19645","snapshot_observed_at":"2026-08-16T00:36:56.151108Z","title":"Fine-tuning vision-language-action models: Optimizing speed and success.arXiv preprint arXiv:2502.19645, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.151108Z"},"links":{"cited_paper":"/paper/2502.19645","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:7cfeb3ccd019debec2eb5e1f286c0f3dbfa5831c62a096a25deeec43eed9e428","observation_id":"82e07ac9-34f3-4fcf-9b67-e54122a22b80","resolution":{"observed_at":"2026-08-16T00:36:56.151108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.16163","last_updated":"2026-01-22T18:09:30Z","snapshot_observed_at":"2026-08-20T07:20:19.015572Z","submitted_at":"2026-01-22T18:09:30Z","title":"Cosmos Policy: Fine-Tuning Video Models for Visuomotor Control and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.16163","snapshot_observed_at":"2026-08-16T00:36:56.155511Z","title":"Cosmos policy: Fine-tuning video models for visuomotor control and planning.arXiv preprint arXiv:2601.16163, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.155511Z"},"links":{"cited_paper":"/paper/2601.16163","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:34f64d46db67fd27dbf8591d143c74ae299d81786da2d0f3e3e8da310db8bd3d","observation_id":"29de1143-8027-4771-b45c-c47939b9b0d3","resolution":{"observed_at":"2026-08-16T00:36:56.155511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.160221Z","title":"Task adaptation of vision-language-action model: 1st place solution for the 2025 behavior challenge.arXiv preprint arXiv:2512.06951, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.160221Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:3049a8dc6eed9899d6324771d0b79cb250a4ca088b661769d847e8b064f4d825","observation_id":"b08c06af-78f7-4153-ab1d-7dcb50220005","resolution":{"observed_at":"2026-08-16T00:36:56.160221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.164151Z","title":"Openpi comet: Competition solution for 2025 behavior challenge.arXiv preprint arXiv:2512.10071, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.164151Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:f887b4c2442013f6ab4498fee1683a844f2eba1efdfa733329d88fd0197640a4","observation_id":"e2785799-22a6-40bd-99c0-33d023b9dab1","resolution":{"observed_at":"2026-08-16T00:36:56.164151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-16T00:36:56.168485Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models.arXiv preprint arXiv:2402.03300, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.168485Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:8f767d154d81289da074d11aadf60b819708eb6b8e6fe2457e4a7440e349c89a","observation_id":"86aa10a3-011e-440d-8c3d-553e90849c7e","resolution":{"observed_at":"2026-08-16T00:36:56.168485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:36:56.172379Z","title":"RLinf: Flexible and efficient large-scale reinforcement learning via macro-to-micro flow transformation.arXiv preprint arXiv:2509.15965, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-16T00:36:56.172379Z"},"links":{"citing_paper":"/paper/2608.11739"},"observation_digest":"sha256:7c13ff2d3650183453b2abb2aa629f1407b60104ec221a34b4ab6700b33bc3f0","observation_id":"b55cee17-9784-4f27-9e01-bc25f179af50","resolution":{"observed_at":"2026-08-16T00:36:56.172379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.11739","last_updated":"2026-08-12T07:26:47Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-18T22:03:23.597514Z","submitted_at":"2026-08-12T07:26:47Z","title":"G0.5: One Autoregressive Stream for Robot Reasoning and Action"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":47,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 0 inbound Pith citation observations for arXiv:2608.11739."}