{"as_of":"2026-08-08T01:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:41d6eb4f776aa3dddded1be25e63c3e5dc794dd801d1014dd3a42d2c78b08bee","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T00:57:45.185859Z","state":"measured"},{"denominator":59,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":59,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.14852/citation-record","integrity":"/paper/2607.14852/integrity","json":"/paper/2607.14852/citation-record.json","paper":"/paper/2607.14852"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2204.01691","last_updated":"2022-08-16T16:06:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-04T17:57:11Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.01691","snapshot_observed_at":"2026-08-02T00:57:38.361430Z","title":"Do as i can, not as i say: Grounding language in robotic affordances.arXiv preprint arXiv:2204.01691, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.361430Z"},"links":{"cited_paper":"/paper/2204.01691","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:76aa84d0a5a17cbc04732f628780208033b6a8790cc6a8588f8721b000001266","observation_id":"6d4e5abc-5875-460f-9143-60c6687f9e05","resolution":{"observed_at":"2026-08-02T00:57:38.361430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-02T00:57:38.446664Z","title":"org/abs/2410.24164, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.446664Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:00df0cc8f9851d27575cde14d2a117fe7518a0674459b486b122b009b5be1d1d","observation_id":"8da652d4-1b83-4fb1-b00e-29b2ac5e0a2e","resolution":{"observed_at":"2026-08-02T00:57:38.446664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11706","last_updated":"2023-12-22T13:55:42Z","snapshot_observed_at":"2026-07-06T15:44:42.776459Z","submitted_at":"2023-06-20T17:35:20Z","title":"RoboCat: A Self-Improving Generalist Agent for Robotic Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11706","snapshot_observed_at":"2026-08-02T00:57:38.564461Z","title":"Robocat: A self-improving generalist agent for robotic manipulation.arXiv preprint arXiv:2306.11706, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.564461Z"},"links":{"cited_paper":"/paper/2306.11706","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:3acbe7be1d38af0b093e5b542e28b5d71d8d2f2b1b6047b814306d43e0d6319c","observation_id":"5ecd9849-3188-4f1e-872d-c057bfe1feaf","resolution":{"observed_at":"2026-08-02T00:57:38.564461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-02T00:57:38.675970Z","title":"Rt-1: Robotics transformer for real-world control at scale.arXiv preprint arXiv:2212.06817, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.675970Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:83ac78fc1e7f5714cf76e7ef23b605d40ef4d1968d0f1f65c5116439060fe2a3","observation_id":"d23d7b75-13be-4f71-b49f-811a87159e0e","resolution":{"observed_at":"2026-08-02T00:57:38.675970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-02T00:57:38.798291Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control, 2023.URL https://arxiv","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.798291Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:790f5a34dc46373c5e1cdb72073756b23958c5c27dd51d5dcd5108e5f7c2c51b","observation_id":"bfd3f953-bb97-48a6-8ad8-0565fa985ce2","resolution":{"observed_at":"2026-08-02T00:57:38.798291Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:38.868336Z","title":"Riemannian walk for incremental learning: Understanding forgetting and intransigence","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.868336Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:ab602a7d68af8e138961497111c787683b72385cd65bb7ed9cc5cbcd43f90194","observation_id":"8fa60eea-fa48-4433-b372-6b24d4ce6af2","resolution":{"observed_at":"2026-08-02T00:57:38.868336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1902.10486","last_updated":"2019-06-04T07:59:35Z","snapshot_observed_at":"2026-08-07T08:44:38.959537Z","submitted_at":"2019-02-27T12:34:19Z","title":"On Tiny Episodic Memories in Continual Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1902.10486","snapshot_observed_at":"2026-08-02T00:57:38.974751Z","title":"On tiny episodic memories in continual learning.arXiv preprint arXiv:1902.10486, 2019","venue":null,"work_id":null,"year":1902},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:38.974751Z"},"links":{"cited_paper":"/paper/1902.10486","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:e37964a0310d773c91d2e0d7788b9feffdd6dd4518c1986296ef9fe874880c7b","observation_id":"ac24ec8c-df59-44f7-9361-66b1e17324df","resolution":{"observed_at":"2026-08-02T00:57:38.974751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.04137","last_updated":"2024-03-14T04:36:31Z","snapshot_observed_at":"2026-08-03T00:02:57.373313Z","submitted_at":"2023-03-07T18:50:03Z","title":"Diffusion Policy: Visuomotor Policy Learning via Action Diffusion","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.04137","snapshot_observed_at":"2026-08-02T00:57:39.140456Z","title":"Diffusion policy: Visuomotor policy learning via action diffusion.arXiv preprint arXiv:2303.04137, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.140456Z"},"links":{"cited_paper":"/paper/2303.04137","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:664a45f6446cc7191e8d9b5dc671bbed829188d459855e9ac8cafe99f98cdbac","observation_id":"97243f6c-be38-4b5a-a6fe-ca2c7dec90b3","resolution":{"observed_at":"2026-08-02T00:57:39.140456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03912","last_updated":"2025-05-06T18:35:07Z","snapshot_observed_at":"2026-08-07T15:49:25.272514Z","submitted_at":"2025-05-06T18:35:07Z","title":"OpenHelix: A Short Survey, Empirical Analysis, and Open-Source Dual-System VLA Model for Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.03912","snapshot_observed_at":"2026-08-02T00:57:39.308136Z","title":"Openhelix: A short survey, empirical analysis, and open-source dual-system vla model for robotic manipulation.arXiv preprint arXiv:2505.03912, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.308136Z"},"links":{"cited_paper":"/paper/2505.03912","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:7dcc8d20df20ff0aef6bbe2bc3f9f3150391e3144d8c006a88e33717526b7b72","observation_id":"cf28c9fc-2b4b-4556-8395-f2bd1a2639dc","resolution":{"observed_at":"2026-08-02T00:57:39.308136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:39.473600Z","title":"Loss of plasticity in deep continual learning.Nature, 632(8026):768–774, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.473600Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:180dbe6e7af328326a71c76257809a13fd403062435f26dd3d8f79b97859430c","observation_id":"1a53177d-b354-436d-bfe3-879228996277","resolution":{"observed_at":"2026-08-02T00:57:39.473600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:39.612805Z","title":"Palm-e: An embodied multimodal language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.612805Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:0265a1b7082e8d6780bf9e3404d7e96b4c0682a4c0d77f66bffd471c3139ae5c","observation_id":"6c03f0c3-b7d1-4ecc-85d8-62fe8f30faf9","resolution":{"observed_at":"2026-08-02T00:57:39.612805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05540","last_updated":"2025-06-17T03:16:46Z","snapshot_observed_at":"2026-08-07T15:48:43.434747Z","submitted_at":"2025-05-08T16:51:36Z","title":"Benchmarking Vision, Language, & Action Models in Procedurally Generated, Open Ended Action Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.05540","snapshot_observed_at":"2026-08-02T00:57:39.775665Z","title":"Benchmarking vision, language, & action models in procedurally generated, open ended action environ- ments.arXiv preprint arXiv:2505.05540, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.775665Z"},"links":{"cited_paper":"/paper/2505.05540","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:cee1b8827c4c1d92c4679713208ff8fef9243233aaba9073896abfb154715514","observation_id":"81402915-0aa4-4d73-9b3f-e82aec51676b","resolution":{"observed_at":"2026-08-02T00:57:39.775665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:39.937349Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:39.937349Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:b252484193c3b602520a289e7ce67df5ab5246b48bb672b566e1521594b5dcc2","observation_id":"4ba87ad8-c25a-46fe-ae7d-0666ed9f4d4b","resolution":{"observed_at":"2026-08-02T00:57:39.937349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.05973","last_updated":"2023-11-02T06:53:37Z","snapshot_observed_at":"2026-08-05T01:03:23.456778Z","submitted_at":"2023-07-12T07:40:48Z","title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.05973","snapshot_observed_at":"2026-08-02T00:57:40.106650Z","title":"Voxposer: Compos- able 3d value maps for robotic manipulation with language models.arXiv preprint arXiv:2307.05973, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.106650Z"},"links":{"cited_paper":"/paper/2307.05973","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:ede741a7c95cbc46e343a2be8c7d5f3444b627d14b85431ce650c28d04f8efda","observation_id":"e5a8d7e6-4f24-4cf7-b19c-973217c16924","resolution":{"observed_at":"2026-08-02T00:57:40.106650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16054","last_updated":"2025-04-22T17:31:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:31:29Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16054","snapshot_observed_at":"2026-08-02T00:57:40.272666Z","title":"5: A vision-language-action model with open-world generalization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.272666Z"},"links":{"cited_paper":"/paper/2504.16054","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:bf7729027ef5571061040b602d7e33e109e8c95db1a51db346e7f4407ca5013f","observation_id":"cae95d81-6b39-474d-a382-5de0ee2bbb6e","resolution":{"observed_at":"2026-08-02T00:57:40.272666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03094","last_updated":"2023-05-28T07:32:38Z","snapshot_observed_at":"2026-08-07T13:44:24.778030Z","submitted_at":"2022-10-06T17:50:11Z","title":"VIMA: General Robot Manipulation with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03094","snapshot_observed_at":"2026-08-02T00:57:40.434766Z","title":"Vima: General robot manipulation with multimodal prompts.arXiv preprint arXiv:2210.03094, 2(3):6, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.434766Z"},"links":{"cited_paper":"/paper/2210.03094","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:3c4ac2d65b3637ca1bdbe4a06f445ef75310f3a7a70db541d5ed7401f0295abd","observation_id":"e7c4049a-4dde-4140-8e42-6f4f21c82016","resolution":{"observed_at":"2026-08-02T00:57:40.434766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:40.557073Z","title":"Srt-h: A hierarchical framework for autonomous surgery via language-conditioned imitation learning.Science robotics, 10(104):eadt5254, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.557073Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:26445472893d989aede4da4a2b9c22ea5ef3c42bf73d3dfab7388d58ff037300","observation_id":"671d6ba0-d7df-499c-9274-8b1d699d88fd","resolution":{"observed_at":"2026-08-02T00:57:40.557073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-02T00:57:40.564504Z","title":"Openvla: An open-source vision-language- action model.arXiv preprint arXiv:2406.09246, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.564504Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:0bafb6e3e0ff39fec99771e335be1f36af00bd8030ed9d0583b50d66a8712bf9","observation_id":"17855fb2-1f2f-45cd-96f9-767e61ea49ca","resolution":{"observed_at":"2026-08-02T00:57:40.564504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:40.626531Z","title":"Overcoming catastrophic forgetting in neural networks.Proceedings of the national academy of sciences, 114(13): 3521–3526, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.626531Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2608ab6840c75e3b178ff2020315568ce18c6a48ba4edf5c5ebf04870151e135","observation_id":"7f1747ac-9dad-42dc-b167-da8070caa109","resolution":{"observed_at":"2026-08-02T00:57:40.626531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.05386","last_updated":"2026-06-29T17:47:49Z","snapshot_observed_at":"2026-08-06T19:25:14.929252Z","submitted_at":"2025-07-07T18:17:06Z","title":"Reinforcement Fine-Tuning Naturally Mitigates Forgetting in Continual Post-Training","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.05386","snapshot_observed_at":"2026-08-02T00:57:40.746075Z","title":"Reinforcement fine-tuning naturally mitigates forgetting in continual post-training.arXiv preprint arXiv:2507.05386, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.746075Z"},"links":{"cited_paper":"/paper/2507.05386","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:b8abc81ea739b2804971aff01eb9f553c943add477e927fe44ba7cbebe5e2bcd","observation_id":"973ce0bb-7239-40ee-8772-a2cee4eca53c","resolution":{"observed_at":"2026-08-02T00:57:40.746075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:40.923216Z","title":"The power of scale for parameter-efficient prompt tuning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:40.923216Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:63c6e2d0de6661eb59cf6504964d3efe5e7e859e2cc1868413bed3305a3e0a42","observation_id":"920df1b8-d243-4890-b5e7-f1a1efd5aa4b","resolution":{"observed_at":"2026-08-02T00:57:40.923216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.063842Z","title":"Remem-vla: Empowering vision-language-action model with memory via dual-level recurrent queries.arXiv preprint arXiv:2603.12942, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.063842Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2673a93779178dec942f7568c166b5db3540d9b7a80882f7e22a37c1e5d76546","observation_id":"1c832505-4417-4d0b-ab31-5a3909d5fd35","resolution":{"observed_at":"2026-08-02T00:57:41.063842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.166042Z","title":"Prefix-tuning: Optimizing continuous prompts for generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.166042Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:4a1e90499729214a6db18bd93e10a592aeef6e6f61764bbfdef68342bcee660f","observation_id":"1db9480a-222b-4401-a7ef-7963e83fa241","resolution":{"observed_at":"2026-08-02T00:57:41.166042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.271781Z","title":"Learning without forgetting.IEEE transactions on pattern analysis and machine intelligence, 40(12):2935–2947, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.271781Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:f9612c65acc7bcee2aacf15665b920f88a0d82e4b8c39789c161452b52a9d700","observation_id":"df862715-64fe-47c8-a6f2-2a3e0c02350e","resolution":{"observed_at":"2026-08-02T00:57:41.271781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.376843Z","title":"Learning without forgetting","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.376843Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:6c66a6bd4d1580da41d3d099c3dc263f30e7941db59fb442fe254f34387e96ea","observation_id":"6d394b78-c309-43d3-aaa3-a0f41b8b07f1","resolution":{"observed_at":"2026-08-02T00:57:41.376843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.519701Z","title":"Code as policies: Language model programs for embodied control","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.519701Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:f664ab6e611d28115391ebcd82736c2e3b3f91220481dff3f27cd7a9ed99eed5","observation_id":"d083236b-4763-4028-b504-68e83c95dba4","resolution":{"observed_at":"2026-08-02T00:57:41.519701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.679900Z","title":"Never-ending behavior-cloning agent for robotic manipulation.arXiv preprint arXiv:2403.00336, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.679900Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2c3d131200066a6027cda5b2cb29ff9e782d60d8b507450cb61eeb89bca6a0e5","observation_id":"8a6204cf-c9a6-406a-81ed-fce26d6c0c00","resolution":{"observed_at":"2026-08-02T00:57:41.679900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:41.790039Z","title":"Pixelvla: Advancing pixel-level understanding in vision-language-action model.arXiv preprint arXiv:2511.01571, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.790039Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:10d4efe009ab7208d92ef312415f6617b8fcbea97725779927917fd6bed50556","observation_id":"ae038217-681c-40b4-934c-8fbd01cfe4a2","resolution":{"observed_at":"2026-08-02T00:57:41.790039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.20072","last_updated":"2026-05-31T15:50:43Z","snapshot_observed_at":"2026-08-05T15:11:00.360328Z","submitted_at":"2025-08-27T17:39:11Z","title":"Discrete Diffusion VLA: Bringing Discrete Diffusion to Action Decoding in Vision-Language-Action Policies","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.20072","snapshot_observed_at":"2026-08-02T00:57:41.909520Z","title":"Discrete diffusion vla: Bringing discrete diffusion to action decoding in vision-language-action policies.arXiv preprint arXiv:2508.20072, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:41.909520Z"},"links":{"cited_paper":"/paper/2508.20072","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:c99856c030ebc880526ed8f2a41f018bd22625f62dede8a5e90994c6ca24b117","observation_id":"f4e12658-d377-4652-acaf-556b08b76456","resolution":{"observed_at":"2026-08-02T00:57:41.909520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.023492Z","title":"Showui: One vision-language-action model for gui visual agent","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.023492Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:14067f529f016efdf19032e22b86ba2eefc347f49c28b803992670a567c4a462","observation_id":"75d14dfb-e1ee-489d-903a-193620f4e0f4","resolution":{"observed_at":"2026-08-02T00:57:42.023492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.163804Z","title":"Libero: Benchmarking knowledge transfer for lifelong robot learning.Advances in Neural Information Processing Systems, 36:44776–44791, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.163804Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:03601e944162b0cd1d4e8000144743aee426c07d6457f2b377fb2f0bb35af8c4","observation_id":"c9dda9e5-a346-4835-9596-a5462827eb2a","resolution":{"observed_at":"2026-08-02T00:57:42.163804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.309881Z","title":"Pretrained vision-language-action models are surprisingly resistant to forgetting in continual learning.arXiv preprint arXiv:2603.03818, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.309881Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:f800e203e9c50f4a364db38c72bc32580bf8e7a6beb7336110f41c76ddfba51a","observation_id":"2861effb-1148-4105-b9e3-6352c2a770b6","resolution":{"observed_at":"2026-08-02T00:57:42.309881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06710","last_updated":"2025-07-13T06:32:40Z","snapshot_observed_at":"2026-08-06T18:54:09.737742Z","submitted_at":"2025-07-09T10:08:15Z","title":"Spatial-Temporal Aware Visuomotor Diffusion Policy Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06710","snapshot_observed_at":"2026-08-02T00:57:42.384758Z","title":"Spatial- temporal aware visuomotor diffusion policy learning.arXiv preprint arXiv:2507.06710, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.384758Z"},"links":{"cited_paper":"/paper/2507.06710","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2fae186cdde2c4e86a65701c69f07b1c941152a9666363d4d6ba45257355fd12","observation_id":"ad0e677d-ee68-4ae8-b596-9ab348fca764","resolution":{"observed_at":"2026-08-02T00:57:42.384758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.499558Z","title":"Packnet: Adding multiple tasks to a single network by iterative pruning","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.499558Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:7b9bf969fe057d9f726a038bfdd08b7b1571f5cfaebc9a602dc6a6ee519c9734","observation_id":"76a1a350-ba0f-4403-b458-12b73c023994","resolution":{"observed_at":"2026-08-02T00:57:42.499558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:42.643648Z","title":"Preserving and combining knowledge in robotic lifelong reinforcement learning.Nature Machine Intelligence, 7(2):256–269, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.643648Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:8813de8d74b6e5cfc9b98fcd42d1331c03efc5fa04c4c7b7a4c26f47373f9084","observation_id":"f859b398-a502-45a9-b0de-7c514b6607a0","resolution":{"observed_at":"2026-08-02T00:57:42.643648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11815","last_updated":"2024-06-17T17:55:29Z","snapshot_observed_at":"2026-07-06T18:32:23.999019Z","submitted_at":"2024-06-17T17:55:29Z","title":"LLARVA: Vision-Action Instruction Tuning Enhances Robot Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11815","snapshot_observed_at":"2026-08-02T00:57:42.768376Z","title":"Llarva: Vision-action instruction tuning enhances robot learning.arXiv preprint arXiv:2406.11815, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.768376Z"},"links":{"cited_paper":"/paper/2406.11815","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:aff523bc4139f344b60a7db10589d8482daa27cfc528f6e5628a14c59753ba21","observation_id":"e21013a9-57c6-4ceb-8be4-35d30bdc961b","resolution":{"observed_at":"2026-08-02T00:57:42.768376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08864","last_updated":"2025-05-14T15:22:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-13T05:20:40Z","title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08864","snapshot_observed_at":"2026-08-02T00:57:42.876938Z","title":"Open x-embodiment: Robotic learning datasets and rt-x models.arXiv preprint arXiv:2310.08864, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.876938Z"},"links":{"cited_paper":"/paper/2310.08864","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:5c1b1e3efe5dc5246dc8e3563a3e4c6dcb2573f756c8223b0f73f28471d7c910","observation_id":"b7e19fac-c9c1-4f25-801d-4be669680dc4","resolution":{"observed_at":"2026-08-02T00:57:42.876938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09747","last_updated":"2025-01-16T18:57:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T18:57:04Z","title":"FAST: Efficient Action Tokenization for Vision-Language-Action Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09747","snapshot_observed_at":"2026-08-02T00:57:42.983225Z","title":"Fast: Efficient action tokenization for vision-language-action models.arXiv preprint arXiv:2501.09747, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:42.983225Z"},"links":{"cited_paper":"/paper/2501.09747","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:5fe0664d7f0f9c7ceb33f77fe94c8578f5c69ff32510505b1a7a625d41d593e4","observation_id":"8e5de4bf-5370-457c-bf11-394968808e29","resolution":{"observed_at":"2026-08-02T00:57:42.983225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15483","last_updated":"2026-04-24T23:18:28Z","snapshot_observed_at":"2026-08-05T18:18:04.172934Z","submitted_at":"2026-04-16T19:18:07Z","title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.15483","snapshot_observed_at":"2026-08-02T00:57:43.044566Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.044566Z"},"links":{"cited_paper":"/paper/2604.15483","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:7adaa583c494f18ad9465fe0f89f097bb8195b6baa9603da19d1f3f6bb95fc5a","observation_id":"f093673d-4b63-4894-9d1d-2b3ffc21433c","resolution":{"observed_at":"2026-08-02T00:57:43.044566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.119826Z","title":"Incremental classifier and representation learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.119826Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:b79f9345fed0f09fbbf7ece6cff24f199de68911512fdfaad788230ab5d1f2d6","observation_id":"deffecab-8279-43d4-a60f-555ffe976a3b","resolution":{"observed_at":"2026-08-02T00:57:43.119826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.06175","last_updated":"2022-11-11T10:04:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-05-12T16:03:26Z","title":"A Generalist Agent","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.06175","snapshot_observed_at":"2026-08-02T00:57:43.200599Z","title":"A generalist agent.arXiv preprint arXiv:2205.06175, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.200599Z"},"links":{"cited_paper":"/paper/2205.06175","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:6943f8177e333ece2bbac2563e48f3e955a570c0231fe5378d5d67c56e5dc270","observation_id":"86645130-7d20-4dd4-838d-caa7f7cfebf8","resolution":{"observed_at":"2026-08-02T00:57:43.200599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.04671","last_updated":"2022-10-22T14:34:44Z","snapshot_observed_at":"2026-08-05T07:46:46.355580Z","submitted_at":"2016-06-15T08:20:51Z","title":"Progressive Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.04671","snapshot_observed_at":"2026-08-02T00:57:43.302391Z","title":"Progressive neural networks.arXiv preprint arXiv:1606.04671, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.302391Z"},"links":{"cited_paper":"/paper/1606.04671","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:2be89b147bbf0f37db628a25842488c3a64d7cab3c4c7078fad775eabd83bf8b","observation_id":"e9c286bb-d113-4d4a-bf70-b76e4fcdae19","resolution":{"observed_at":"2026-08-02T00:57:43.302391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.04259","last_updated":"2025-09-04T14:38:08Z","snapshot_observed_at":"2026-08-07T12:32:15.780660Z","submitted_at":"2025-09-04T14:38:08Z","title":"RL's Razor: Why Online Reinforcement Learning Forgets Less","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.04259","snapshot_observed_at":"2026-08-02T00:57:43.368877Z","title":"Rl’s razor: Why online reinforcement learning forgets less.arXiv preprint arXiv:2509.04259, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.368877Z"},"links":{"cited_paper":"/paper/2509.04259","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:c0563ff90bb943e8ef60339301f84e202844ec31447ad5d659580e0a92b9ec4b","observation_id":"1d130e34-7e7e-4643-96c8-fc4b7eb9824f","resolution":{"observed_at":"2026-08-02T00:57:43.368877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.430502Z","title":"Cliport: What and where pathways for robotic manipulation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.430502Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:a13f7b02345182090b4d362bd313e31822ab9804af388722dff2d94af8d7b4a0","observation_id":"4091b805-2fee-4570-bc78-e314dad59af3","resolution":{"observed_at":"2026-08-02T00:57:43.430502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.556355Z","title":"Perceiver-actor: A multi-task transformer for robotic manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.556355Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:47ef43ffbaa1b96cf49bea29666b9361e875797cfb214bab16a61a380874a655","observation_id":"30b3ec67-9175-4544-8a41-6fa4ed8db12b","resolution":{"observed_at":"2026-08-02T00:57:43.556355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01844","last_updated":"2025-06-02T16:30:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T16:30:19Z","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01844","snapshot_observed_at":"2026-08-02T00:57:43.651170Z","title":"Smolvla: A vision-language- action model for affordable and efficient robotics.arXiv preprint arXiv:2506.01844, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.651170Z"},"links":{"cited_paper":"/paper/2506.01844","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:b32bd6ba5d393040f8fe1e4d11807f4aa88a67576e52ef2cbede237844adb8cd","observation_id":"2b2a4d7c-00e1-47e2-a65c-e5d731c23ac4","resolution":{"observed_at":"2026-08-02T00:57:43.651170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:43.723645Z","title":"Coda-prompt: Continual decomposed attention- based prompting for rehearsal-free continual learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.723645Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:d677c0a012a2dbbc3938d4fde2ec164c78bcf2e01477a61f19cad6ba731e67db","observation_id":"68528695-96c2-4fa0-a50c-627e6815c1ab","resolution":{"observed_at":"2026-08-02T00:57:43.723645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03189","last_updated":"2025-05-30T20:52:21Z","snapshot_observed_at":"2026-08-07T21:20:32.878637Z","submitted_at":"2025-05-30T20:52:21Z","title":"Continual Learning in Vision-Language Models via Aligned Model Merging","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.03189","snapshot_observed_at":"2026-08-02T00:57:43.881234Z","title":"Continual learning in vision-language models via aligned model merging.arXiv preprint arXiv:2506.03189, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:43.881234Z"},"links":{"cited_paper":"/paper/2506.03189","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:7793b4e5c4f3c1af8d520a8987f9817beefb4ebd79ccd1497d0627743e780e76","observation_id":"58294f10-753f-46e2-895f-e3ea0d455fe2","resolution":{"observed_at":"2026-08-02T00:57:43.881234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12213","last_updated":"2024-05-26T19:55:26Z","snapshot_observed_at":"2026-07-06T18:16:51.116432Z","submitted_at":"2024-05-20T17:57:01Z","title":"Octo: An Open-Source Generalist Robot Policy","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12213","snapshot_observed_at":"2026-08-02T00:57:44.038690Z","title":"Octo: An open-source generalist robot policy.arXiv preprint arXiv:2405.12213, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.038690Z"},"links":{"cited_paper":"/paper/2405.12213","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:6cfcc3a98453378d41b1175e23983e401706dd63702522705f119c7dd785a58d","observation_id":"3449da03-ae32-4162-bdf3-0cbecd78b540","resolution":{"observed_at":"2026-08-02T00:57:44.038690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.188337Z","title":"A comprehensive survey of continual learning: Theory, method and application.IEEE transactions on pattern analysis and machine intelligence, 46(8): 5362–5383, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.188337Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:ad98757e82c4a39a2520709a81d6b6f431b408440f211dd8cd9833b43430b7db","observation_id":"a5dcc6ca-0af3-4495-8190-012c45d24af2","resolution":{"observed_at":"2026-08-02T00:57:44.188337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.04799","last_updated":"2022-08-05T11:26:06Z","snapshot_observed_at":"2026-08-07T01:22:45.613150Z","submitted_at":"2022-04-10T23:36:55Z","title":"DualPrompt: Complementary Prompting for Rehearsal-free Continual Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.04799","snapshot_observed_at":"2026-08-02T00:57:44.354469Z","title":"Dualprompt: Complementary prompting for rehearsal-free continual learning.arXiv preprint arXiv:2204.04799, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.354469Z"},"links":{"cited_paper":"/paper/2204.04799","citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:1d5ae7f9bbbcf4761a8d0ead50df743eac08058fcddae017f36740e78055bae0","observation_id":"697e9aba-0960-4c83-8869-10f31a36f1bd","resolution":{"observed_at":"2026-08-02T00:57:44.354469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.474480Z","title":"Tinyvla: Towards fast, data-efficient vision-language-action models for robotic manipulation.IEEE Robotics and Automation Letters, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.474480Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:caa2b2bfd1f694d5b1e0b0e311df4dd8c20e8f4707c4b3cd8e998e96bf133163","observation_id":"c29e4a62-2969-42e5-b3e9-618575a75da8","resolution":{"observed_at":"2026-08-02T00:57:44.474480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.733577Z","title":"Continual world: A robotic benchmark for continual reinforcement learning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.733577Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:1d499f98671bc9dd5711c37b31a01a3a355d81d2e098b9eaad31ab430905c471","observation_id":"69001641-1349-411b-8022-4794ddb7017b","resolution":{"observed_at":"2026-08-02T00:57:44.733577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.809271Z","title":"Long-horizon language-conditioned imitation learning for robotic manipulation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.809271Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:9e30afd7e23976e85c8bdabcfe6a5dc7da2aebeefe7d28d156b98c87d877e9e7","observation_id":"543c0b8f-1bf6-4ffa-b97d-a34062faa093","resolution":{"observed_at":"2026-08-02T00:57:44.809271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:44.957923Z","title":"Boosting continual learning of vision-language models via mixture-of-experts adapters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:44.957923Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:6711a92ab3c7e01cc5ae1addb8f1d08a6bf2b47c9d8f0a30825a40c7533a9b20","observation_id":"9c3e7bb1-682a-4721-b50b-ea52e0df196e","resolution":{"observed_at":"2026-08-02T00:57:44.957923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.006248Z","title":"Atomicvla: Unlocking the potential of atomic skill learning in robots.arXiv preprint arXiv:2603.07648, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.006248Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:7d952d4a8ebfa15a994584178451f970fd043f3c5c2132294055439aa5d66fa4","observation_id":"6a83ff64-d458-4670-9690-e4b773621fde","resolution":{"observed_at":"2026-08-02T00:57:45.006248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.048809Z","title":"Mllm-cl: Continual learning for multimodal large language models.arXiv preprint arXiv:2506.05453, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.048809Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:15ac8785eb1b16765f3b5f9fdc6fedc0174c2e9a56631d5a34e148b149ea7df2","observation_id":"20ba1cbf-ecb0-48d1-9ff7-2422168cd8fa","resolution":{"observed_at":"2026-08-02T00:57:45.048809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.102602Z","title":"Information-theoretic constraints for continual vision-language-action alignment.arXiv preprint arXiv:2603.13335, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.102602Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:b521c6be658a499de959356c7f32dfb9fe197338326ec821aa7e7f8af0653b95","observation_id":"3bf46d08-2adf-45ed-b526-6ccba4889e8b","resolution":{"observed_at":"2026-08-02T00:57:45.102602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T00:57:45.185859Z","title":"imanip: Skill-incremental learning for robotic manipulation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-02T00:57:45.185859Z"},"links":{"citing_paper":"/paper/2607.14852"},"observation_digest":"sha256:f5adc46baa6ab20aa253fc3738ca002acb28fd8a0fbd3fbd260dea5a7c56153d","observation_id":"965b4692-6aed-46b2-801c-c004dcfbc883","resolution":{"observed_at":"2026-08-02T00:57:45.185859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.14852","last_updated":"2026-07-21T06:49:06Z","latest_version":2,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-02T04:15:16.437644Z","submitted_at":"2026-07-16T11:22:06Z","title":"Towards Human-like Physical Intelligence: Lifelong Vision-Language-Action Learning for Robotic Manipulation"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":59,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 0 inbound Pith citation observations for arXiv:2607.14852."}