{"as_of":"2026-08-05T15:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:189623e5e8dd8860daebb97907c42fb9fa19dfbc20643536743901b93e9847ae","coverage":[{"denominator":69,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-11T00:33:50.471804Z","state":"measured"},{"denominator":169,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":169,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":376,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T06:04:03.420693Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":3,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2502.02779","last_updated":"2026-04-21T17:21:08Z","snapshot_observed_at":"2026-07-06T20:31:17.809419Z","submitted_at":"2025-02-04T23:42:18Z","title":"3D Foundation Model for Generalizable Disease Detection in Head Computed Tomography","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T03:26:46.665351Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2502.02779"},"observation_digest":"sha256:c64f4c99870ad265496663248e0a289ac309351324ff6127a06bcc3bc35eea68","observation_id":"21a9cdcc-18e3-4368-8bc6-7f934b17a38c","resolution":{"observed_at":"2026-05-23T03:27:27.024372Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2507.01925","last_updated":"2025-07-02T17:34:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:34:52Z","title":"A Survey on Vision-Language-Action Models: An Action Tokenization Perspective","version":1},"reference_index":298,"source":"pdf_text","source_observed_at":"2026-05-17T14:08:34.893876Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2507.01925"},"observation_digest":"sha256:323f96ceec19d9d4e629181e21e00175772364274d29e1dab47fe9e2f75bca4c","observation_id":"3f2e97c4-f009-4f96-bf8a-1d1219edbc22","resolution":{"observed_at":"2026-05-17T14:08:35.462325Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2508.13073","last_updated":"2025-09-01T08:10:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-18T16:45:48Z","title":"Large VLM-based Vision-Language-Action Models for Robotic Manipulation: A Survey","version":2},"reference_index":201,"source":"pdf_text","source_observed_at":"2026-05-17T20:28:15.818016Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2508.13073"},"observation_digest":"sha256:5cbaf706da1c7441712981802b0005bfd6556fbb8ad2f4432e65649f362940e3","observation_id":"61102fb8-a0dc-4cd7-b391-66e53c861e88","resolution":{"observed_at":"2026-05-17T20:28:16.240454Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T06:04:03.420693Z","title":"V-JEPA 2: Self-supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.07996","last_updated":"2026-07-20T17:34:32Z","snapshot_observed_at":"2026-08-05T06:03:57.824747Z","submitted_at":"2025-09-04T17:59:58Z","title":"3D and 4D World Modeling: A Survey","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T06:04:03.420693Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2509.07996"},"observation_digest":"sha256:0e1b0e915c60b1c3524e6e566681acbd65825af3462d3afad6e477b749929b37","observation_id":"d30e72f5-c6d1-4ed0-9cef-06c758ff7752","resolution":{"observed_at":"2026-08-05T06:04:03.420693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:aeb5a334dfcb3b0e9b76cb97951ba377a214eca01b6e8f777a01d5ba5e19efba","observation_id":"d4b14d17-b51d-4562-b0c4-b0b79f98482b","resolution":{"observed_at":"2026-05-18T00:02:24.869240Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2509.20328","last_updated":"2025-09-29T20:44:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-24T17:17:27Z","title":"Video models are zero-shot learners and reasoners","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-14T02:16:45.554252Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2509.20328"},"observation_digest":"sha256:8fcf0527deafc06ed831d0955c4f3e0bf851a4a5745ca3e98613d747db188f8a","observation_id":"96208e04-2720-4f38-9ca4-5006ef5ab42c","resolution":{"observed_at":"2026-05-14T02:16:45.787940Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2509.24948","last_updated":"2026-04-27T05:41:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-29T15:45:19Z","title":"World-Env: Leveraging World Model as a Virtual Environment for VLA Post-Training","version":6},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T12:48:32.123998Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2509.24948"},"observation_digest":"sha256:a05013adeec24e3e0a3f44bafb08c17211be6a8a90bbe7d79685935982f4b146","observation_id":"7d0e224a-43b3-4916-99fb-1379ff8754b7","resolution":{"observed_at":"2026-05-18T12:51:23.537135Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2510.02311","last_updated":"2026-04-13T04:19:16Z","snapshot_observed_at":"2026-08-04T23:56:37.495411Z","submitted_at":"2025-10-02T17:59:50Z","title":"Inferring Dynamic Physical Properties from Video Foundation Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T10:08:11.191706Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2510.02311"},"observation_digest":"sha256:d2818e6494d4d14f119c2137591e2038b95b1613dfd835ed2f36f05e910940af","observation_id":"c0a10ac1-b6f2-4aeb-9670-ff93d76bbf30","resolution":{"observed_at":"2026-05-18T10:11:14.124897Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-04T11:26:04.215319Z","title":"V- JEPA 2: Self - Supervised Video Models Enable Understanding , Prediction and Planning , June 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.04704","last_updated":"2026-05-28T12:47:58Z","snapshot_observed_at":"2026-08-04T11:25:51.992516Z","submitted_at":"2025-10-06T11:17:56Z","title":"AtomWorld: A Benchmark for Evaluating Spatial Reasoning in Large Language Models on Crystalline Materials","version":4},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-04T11:26:04.215319Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2510.04704"},"observation_digest":"sha256:5cdfbb025ba47eb211e766708677e28058415ba3257860d90ed551a341588224","observation_id":"0ce7c558-6d52-4d25-ad06-07093d0bdb01","resolution":{"observed_at":"2026-08-04T11:26:04.215319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2510.04800","last_updated":"2026-04-21T13:16:20Z","snapshot_observed_at":"2026-08-02T20:03:37.497118Z","submitted_at":"2025-10-06T13:30:07Z","title":"Hybrid Architectures for Language Models: Systematic Analysis and Design Insights","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-18T10:18:04.431436Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2510.04800"},"observation_digest":"sha256:16b1a67d60b6cfaac7683625ecb162b2d28388a9df24f49868552b067cd2904b","observation_id":"f2381dac-5508-4b30-95f3-13fea060946d","resolution":{"observed_at":"2026-05-18T10:21:15.006979Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-04T09:12:30.341116Z","title":"V-JEPA 2: Self- supervised video models enable understanding, prediction and planning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.16732","last_updated":"2026-06-25T19:54:02Z","snapshot_observed_at":"2026-08-04T09:12:26.148461Z","submitted_at":"2025-10-19T07:12:32Z","title":"A Comprehensive Survey on World Models for Embodied AI","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T09:12:30.341116Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2510.16732"},"observation_digest":"sha256:bd1056483a35c701b06b4d3ae7b2d97bef9edd60c1df8ed1635eb4b74a56cfdf","observation_id":"38d2428e-0744-43b7-88d2-6041cc7a3d29","resolution":{"observed_at":"2026-08-04T09:12:30.341116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2511.00062","last_updated":"2026-02-24T21:52:50Z","snapshot_observed_at":"2026-07-06T22:34:38.619949Z","submitted_at":"2025-10-28T22:44:13Z","title":"World Simulation with Video Foundation Models for Physical AI","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T23:01:13.546110Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2511.00062"},"observation_digest":"sha256:c6f4f55e9763d982ee21b6a2c4204bbd448131db71783d4db06ae11a8de6f3c1","observation_id":"ab3cb743-68a0-4ace-b948-d3a1d9d89180","resolution":{"observed_at":"2026-05-12T23:01:13.764797Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2511.04670","last_updated":"2025-11-06T18:55:17Z","snapshot_observed_at":"2026-07-06T22:35:07.844255Z","submitted_at":"2025-11-06T18:55:17Z","title":"Cambrian-S: Towards Spatial Supersensing in Video","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T03:46:04.363500Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2511.04670"},"observation_digest":"sha256:0d7264086b26948a34b9fa327506fdba6d78dacb33f7ec18aa60bc2f24cedcad","observation_id":"4e4b2129-5193-46d3-a387-fc376242db98","resolution":{"observed_at":"2026-05-18T03:46:04.539471Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2511.15407","last_updated":"2026-05-24T06:09:04Z","snapshot_observed_at":"2026-08-03T21:29:28.726590Z","submitted_at":"2025-11-19T13:04:44Z","title":"IPR-1: Interactive Physical Reasoner","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T20:55:10.819298Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2511.15407"},"observation_digest":"sha256:d7cf6d1a2260dd5f71ad16412a5416c28b3232e9b0b4effc08a859cb8bb8f068","observation_id":"d7ad238f-d48f-47bf-b56d-6e0de984978b","resolution":{"observed_at":"2026-05-17T20:55:14.957837Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T21:29:31.124396Z","title":"V-jepa 2: Self- supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.15407","last_updated":"2026-05-24T06:09:04Z","snapshot_observed_at":"2026-08-03T21:29:28.726590Z","submitted_at":"2025-11-19T13:04:44Z","title":"IPR-1: Interactive Physical Reasoner","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T21:29:31.124396Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2511.15407"},"observation_digest":"sha256:e21f2e8f1a897669ccf5dba762e9e9d44ccea405952e929c5dd6bbde1462c6bb","observation_id":"dda7fd4f-7d7d-4d7e-a0bc-1b85c5188586","resolution":{"observed_at":"2026-08-03T21:29:31.124396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2511.16567","last_updated":"2026-05-06T15:47:48Z","snapshot_observed_at":"2026-07-06T22:36:30.947164Z","submitted_at":"2025-11-20T17:22:51Z","title":"POMA-3D: The Point Map Way to 3D Scene Understanding","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T20:27:27.347592Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2511.16567"},"observation_digest":"sha256:cc67d53bf0947b23ac30b33cc8bc3d3f4066c6e0d5af98e6b867baadb89b0a65","observation_id":"9d703c97-a3c9-4ed9-8e12-712545cabcff","resolution":{"observed_at":"2026-05-17T20:30:11.562918Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T20:36:05.488757Z","title":"V-jepa 2: Self- supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.19356","last_updated":"2026-07-17T04:00:04Z","snapshot_observed_at":"2026-08-03T20:36:00.140753Z","submitted_at":"2025-11-24T17:56:03Z","title":"Rethinking Reward Signals in Video GRPO: When Scores Become Targets","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T20:36:05.488757Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2511.19356"},"observation_digest":"sha256:2844d2b122e97733d8104d69d41e4f71ae69ee02a53236261de5e5d06c7e7e8b","observation_id":"7a26b75e-ccdb-4541-94ad-957a160d7f01","resolution":{"observed_at":"2026-08-03T20:36:05.488757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T18:09:47.097419Z","title":"V-jepa 2: Self-supervised video models en- able understanding, prediction and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.06628","last_updated":"2026-07-05T11:17:31Z","snapshot_observed_at":"2026-08-03T18:09:44.634080Z","submitted_at":"2025-12-07T02:28:06Z","title":"MIND-V: Hierarchical World Model for Long-Horizon Robotic Manipulation with RL-based Physical Alignment","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T18:09:47.097419Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.06628"},"observation_digest":"sha256:3397b8d2ec1dd85e59b2c2d71936112c82c6b673cb7dbf1d809128187abde113","observation_id":"c58b4f50-b54f-4fa5-b31c-2ea5c5e1e5d0","resolution":{"observed_at":"2026-08-03T18:09:47.097419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T17:52:06.917547Z","title":"V-jepa 2: Self- supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.08029","last_updated":"2026-07-06T17:23:11Z","snapshot_observed_at":"2026-08-03T17:52:05.790846Z","submitted_at":"2025-12-08T20:42:10Z","title":"CLARITY: Medical World Model for Guiding Treatment Decisions by Modeling Context-Aware Disease Trajectories in Latent Space","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T17:52:06.917547Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.08029"},"observation_digest":"sha256:2dc3a6ed12aa87b58e73c9f41147c63dc37e2adc696145789133e0bbf6f87141","observation_id":"adac12f7-08b5-4027-be17-8ce31abf9a29","resolution":{"observed_at":"2026-08-03T17:52:06.917547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T17:09:18.919548Z","title":"V-jepa 2: Self- supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.10607","last_updated":"2026-07-14T22:19:55Z","snapshot_observed_at":"2026-08-03T17:31:19.364600Z","submitted_at":"2025-12-11T13:03:03Z","title":"Track and Caption Any Motion: Open-Vocabulary Spatiotemporal Captioning via Trajectory-Conditioned Generation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T17:09:18.919548Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.10607"},"observation_digest":"sha256:642ef8a4304eb03b68fc0b93114c8c395012e4394e7c9f400266d3b5da62ad6f","observation_id":"aef58132-70af-47e3-981d-a8f1dd0a24c8","resolution":{"observed_at":"2026-08-03T17:09:18.919548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T16:46:58.559497Z","title":"2506.09985","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.12225","last_updated":"2026-06-06T14:32:23Z","snapshot_observed_at":"2026-08-03T16:46:54.598070Z","submitted_at":"2025-12-13T07:39:53Z","title":"A Geometric Theory of Cognition for Machine Intelligence","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T16:46:58.559497Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.12225"},"observation_digest":"sha256:5bf69be7a8b1e84f0f59d57ac3b4d3c46e877e9817a58ade3dff85f3bf455853","observation_id":"31f0bd24-cdaf-4924-a56e-f53b0ba3020d","resolution":{"observed_at":"2026-08-03T16:46:58.559497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2512.13684","last_updated":"2026-04-21T13:13:49Z","snapshot_observed_at":"2026-07-06T22:39:04.164003Z","submitted_at":"2025-12-15T18:59:48Z","title":"Recurrent Video Masked Autoencoders","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T21:55:42.555679Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.13684"},"observation_digest":"sha256:6a8813f6aa7a41e74653f14ba6a251c1b78031f10bed62122e9a4a7bc5763a5b","observation_id":"af32d96c-5b12-46c1-b227-b72cec58107a","resolution":{"observed_at":"2026-05-16T21:58:35.830030Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2512.15692","last_updated":"2025-12-19T18:30:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-17T18:47:31Z","title":"mimic-video: Video-Action Models for Generalizable Robot Control Beyond VLAs","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T10:41:00.142543Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.15692"},"observation_digest":"sha256:32acac6c1759f5354983438e4e23e4dfc4f3a7a56e3040777c52784693f7bea1","observation_id":"3f6f8a27-8e83-43a3-9a5b-74f822c77d77","resolution":{"observed_at":"2026-05-15T10:41:00.299216Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T14:24:49.389952Z","title":"V-JEPA 2: Self-supervised video models enable understanding, predic- tion and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.20606","last_updated":"2026-06-29T06:12:05Z","snapshot_observed_at":"2026-08-03T14:24:48.216973Z","submitted_at":"2025-12-23T18:54:10Z","title":"Probing and Leveraging Video Diffusion Transformer Features for Robust Point Tracking","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T14:24:49.389952Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.20606"},"observation_digest":"sha256:dfa57531b800a00b98361875887ad475d80aba7150eef81965e2153b040d0694","observation_id":"ae34eebe-d67b-4c98-ad3d-87c153291f73","resolution":{"observed_at":"2026-08-03T14:24:49.389952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2512.23421","last_updated":"2026-04-17T11:51:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-29T12:32:27Z","title":"DriveLaW:Unifying Planning and Video Generation in a Latent Driving World","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T19:34:39.518649Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.23421"},"observation_digest":"sha256:5987587f44d734835e726725bd22a94673e044886345afcb00a6c1e02888979d","observation_id":"f9499361-b3e8-4730-a19b-f05269409e1f","resolution":{"observed_at":"2026-05-16T19:38:21.072718Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2512.23864","last_updated":"2026-06-28T00:20:33Z","snapshot_observed_at":"2026-08-03T13:35:53.078536Z","submitted_at":"2025-12-29T21:06:33Z","title":"Learning to Feel the Future: DreamTacVLA for Contact-Rich Manipulation","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T18:39:59.449746Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.23864"},"observation_digest":"sha256:23686132caeae7192392fb039fa34096c74d0e573a98d666b74ea69afb58446f","observation_id":"0fd2bcdf-49c0-47e7-b972-a559bf8f87f7","resolution":{"observed_at":"2026-05-16T18:41:10.949440Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T13:35:54.025798Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.23864","last_updated":"2026-06-28T00:20:33Z","snapshot_observed_at":"2026-08-03T13:35:53.078536Z","submitted_at":"2025-12-29T21:06:33Z","title":"Learning to Feel the Future: DreamTacVLA for Contact-Rich Manipulation","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T13:35:54.025798Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2512.23864"},"observation_digest":"sha256:aaaa145255600c49a58fcf42b06859c9fc6abcb89c07ed75ac4f5a909691a5f9","observation_id":"ad22814e-df61-40a5-9c25-ad49bd2911d4","resolution":{"observed_at":"2026-08-03T13:35:54.025798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T13:00:51.146564Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and planning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.01075","last_updated":"2026-05-28T18:42:26Z","snapshot_observed_at":"2026-08-04T09:14:38.212981Z","submitted_at":"2026-01-03T05:22:27Z","title":"Flow Equivariant World Models: Memory for Partially Observed Dynamic Environments","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-03T13:00:51.146564Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2601.01075"},"observation_digest":"sha256:f311482ee666bd253b63edf6d932459522fccf3b738f0e433c19902ae1ca0267","observation_id":"baa17504-3b25-4f8f-9df5-f436e1ccdb58","resolution":{"observed_at":"2026-08-03T13:00:51.146564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T12:37:27.943411Z","title":"V-JEPA 2: Self-supervised video models enable understanding, prediction and planning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.02295","last_updated":"2026-07-28T17:46:45Z","snapshot_observed_at":"2026-08-03T12:37:21.822048Z","submitted_at":"2026-01-05T17:31:01Z","title":"CycleVLA: Proactive Self-Correcting Vision-Language-Action Models via Subtask Backtracking and Minimum Bayes Risk Decoding","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-03T12:37:27.943411Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2601.02295"},"observation_digest":"sha256:ea525b671a145e981923cc76cb04039bb9ba2cac4f9f52a5e75ec7eba0f25cb9","observation_id":"8bc65591-6b88-4c89-bb64-14a49b14b01e","resolution":{"observed_at":"2026-08-03T12:37:27.943411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2601.20540","last_updated":"2026-01-28T12:37:01Z","snapshot_observed_at":"2026-08-03T01:29:14.204066Z","submitted_at":"2026-01-28T12:37:01Z","title":"Advancing Open-source World Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T09:07:00.904794Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2601.20540"},"observation_digest":"sha256:75dfd5b6c5bd5a8f7a0d233c42cced7b2a0ab43ed406d8f66f7f97db188f11a1","observation_id":"59019341-dc3a-42cf-9eb0-5eb63e765887","resolution":{"observed_at":"2026-05-16T09:07:01.064580Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T06:48:13.911798Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2601.22032","last_updated":"2026-07-02T16:34:59Z","snapshot_observed_at":"2026-08-03T06:48:13.216792Z","submitted_at":"2026-01-29T17:39:20Z","title":"Drive-JEPA: Video JEPA Meets Multimodal Trajectory Distillation for End-to-End Driving","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T06:48:13.911798Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2601.22032"},"observation_digest":"sha256:fd3e3b83382bf35bd606249b5fa148196217ec18de1728081c3ac0a6d3ff5e14","observation_id":"ccbc70af-4af1-4c23-95a0-32fde3d2353b","resolution":{"observed_at":"2026-08-03T06:48:13.911798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.04583","last_updated":"2026-04-17T16:27:47Z","snapshot_observed_at":"2026-07-06T22:44:33.166340Z","submitted_at":"2026-02-04T14:10:36Z","title":"PEPR: Privileged Event-based Predictive Regularization for Domain Generalization","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T07:37:38.263234Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.04583"},"observation_digest":"sha256:e40e3d786752bc8d160a60dbabd0bf7bcdbb630b56c9c28ef086eee2142b1d85","observation_id":"10d87c3c-b641-4255-8841-35c48e2c3b4b","resolution":{"observed_at":"2026-05-16T07:40:44.111305Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.05638","last_updated":"2026-04-17T08:31:14Z","snapshot_observed_at":"2026-08-04T22:30:41.280585Z","submitted_at":"2026-02-05T13:18:33Z","title":"SurgMotion: A Video-Native Foundation Model for Universal Understanding of Surgical Videos","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T07:16:29.588452Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.05638"},"observation_digest":"sha256:2c9c894fd7faee0254205bb8d412683e697c41862ed5b0fd9a27fa123b31ad7b","observation_id":"b5c27669-8051-4f81-9fca-a33b4e8e9d6e","resolution":{"observed_at":"2026-05-16T07:17:30.309716Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.06949","last_updated":"2026-02-06T18:49:43Z","snapshot_observed_at":"2026-07-06T22:44:51.126114Z","submitted_at":"2026-02-06T18:49:43Z","title":"DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T17:02:33.997887Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.06949"},"observation_digest":"sha256:db671867843136cd1c6526c78ee9f735e848b1b02203733a9086d1ee59a6a865","observation_id":"d47e3b7b-3f15-4065-a7fd-c0029e97b160","resolution":{"observed_at":"2026-05-16T17:02:34.074516Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.07064","last_updated":"2026-04-07T13:49:35Z","snapshot_observed_at":"2026-08-03T13:22:02.057620Z","submitted_at":"2026-02-05T14:04:51Z","title":"OmniFysics: Towards Physical Intelligence Evolution via Omni-Modal Signal Processing and Network Optimization","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T07:09:46.254851Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.07064"},"observation_digest":"sha256:2f677144f46d347ba95ee7a92fe472c6f12cdbeb0e6f51db8dadd3c68d144cf6","observation_id":"53dafe14-8e19-4c33-bf29-1a261ef3d584","resolution":{"observed_at":"2026-05-16T07:10:43.176327Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T03:41:53.258454Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning, June 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.07413","last_updated":"2026-06-09T17:59:12Z","snapshot_observed_at":"2026-08-03T03:41:52.703042Z","submitted_at":"2026-02-07T07:18:00Z","title":"Going with the Flow: Koopman Behavioral Models as Pseudo Planners for Visuo-Motor Dexterity","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T03:41:53.258454Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.07413"},"observation_digest":"sha256:074db4ead680acc26fa53f6554862f857885e93602f4bbd59ed9f92382d87418","observation_id":"97ca2aad-0555-4096-b31d-63295199f0b3","resolution":{"observed_at":"2026-08-03T03:41:53.258454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.07775","last_updated":"2026-05-03T01:49:14Z","snapshot_observed_at":"2026-07-30T09:22:37.641917Z","submitted_at":"2026-02-08T02:16:02Z","title":"Rolling Sink: Bridging Limited-Horizon Training and Open-Ended Testing in Autoregressive Video Diffusion","version":6},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T07:02:38.876518Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.07775"},"observation_digest":"sha256:155e363eff7b60cf65fb75dc92b4368b4457cc16029c4b81aa1f20af80e48ee1","observation_id":"8574663c-49b5-495e-ad2d-e4e5c55c2037","resolution":{"observed_at":"2026-05-16T07:07:30.029514Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-03T01:20:04.620117Z","title":"V-JEPA 2: Self-supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.10104","last_updated":"2026-05-26T06:14:19Z","snapshot_observed_at":"2026-08-03T01:20:03.338758Z","submitted_at":"2026-02-10T18:58:41Z","title":"Olaf-World: Orienting Latent Actions for Video World Modeling","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T01:20:04.620117Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.10104"},"observation_digest":"sha256:97dbc49e5eccda46cff3019d6f5f95f9005ff94f8fc3210a69618004220f0fad","observation_id":"612e5dc3-2012-409f-9649-c32baa742f19","resolution":{"observed_at":"2026-08-03T01:20:04.620117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.11075","last_updated":"2026-04-28T08:51:16Z","snapshot_observed_at":"2026-08-03T04:09:34.408025Z","submitted_at":"2026-02-11T17:43:36Z","title":"RISE: Self-Improving Robot Policy with Compositional World Model","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T02:28:37.997148Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.11075"},"observation_digest":"sha256:46293a8172627db9e3cacbfef09900e229f4e3739c0077f39093f685b7d6841b","observation_id":"3cde03a6-daf4-4890-863d-8934ac6bd406","resolution":{"observed_at":"2026-05-16T02:30:32.159613Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.15922","last_updated":"2026-02-17T15:04:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-17T15:04:02Z","title":"World Action Models are Zero-shot Policies","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T16:18:15.003371Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.15922"},"observation_digest":"sha256:3306f871b11d8e9b53c871e834964e0f6a6d414a97da38293d7feef8d1ee8aa9","observation_id":"edea824e-f551-4587-8b00-c2326bdfcfe1","resolution":{"observed_at":"2026-05-11T16:18:15.244552Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T22:25:57.827119Z","title":"Hangbo Bao, Li Dong, and Furu Wei","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.16918","last_updated":"2026-07-13T23:32:16Z","snapshot_observed_at":"2026-08-02T22:25:56.850881Z","submitted_at":"2026-02-18T22:22:44Z","title":"Xray-Visual Models: Scaling Vision models on Industry Scale Data","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T22:25:57.827119Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.16918"},"observation_digest":"sha256:37c2529ff639f702a82a20f0fad8f785eaa9ccebac1cb499e12cffb9f5b57baa","observation_id":"8ca2d4af-bf68-4a7a-b8bf-f51fae9b828e","resolution":{"observed_at":"2026-08-02T22:25:57.827119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.21668","last_updated":"2026-05-04T08:33:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-25T08:04:07Z","title":"Space-Time Forecasting of Dynamic Scenes with Motion-aware Gaussian Grouping","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T19:51:58.756605Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.21668"},"observation_digest":"sha256:fb2bbd5dfbf4f81b9869cae91bf38eacc08c7966cd616b5b0607bed4beee7847","observation_id":"8c0c1ec0-bcc9-43cf-814f-8a6d3b71c92b","resolution":{"observed_at":"2026-05-15T19:56:33.994859Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.22779","last_updated":"2026-06-03T09:11:47Z","snapshot_observed_at":"2026-08-02T20:38:33.149530Z","submitted_at":"2026-02-26T09:15:34Z","title":"TrajTok: Learning Trajectory Tokens enables better Video Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T19:11:52.694778Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.22779"},"observation_digest":"sha256:4eadef44c3606c109b3f04830e4c3a52285b3728c2194cc7314eae202517d0a6","observation_id":"6276ba32-2cb3-4495-86f7-5f86d7344504","resolution":{"observed_at":"2026-05-15T19:16:31.936451Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T20:38:35.343597Z","title":"V-jepa 2: Self-supervised video models enable understanding, predic- tion and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.22779","last_updated":"2026-06-03T09:11:47Z","snapshot_observed_at":"2026-08-02T20:38:33.149530Z","submitted_at":"2026-02-26T09:15:34Z","title":"TrajTok: Learning Trajectory Tokens enables better Video Understanding","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T20:38:35.343597Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.22779"},"observation_digest":"sha256:7d1d5e6495a677dbfb214bfc5f6fd3aaa5fc8e0b97a263cf176d79751a8e9e29","observation_id":"abbaeea2-1fbe-4f5e-a560-ab79d31205d2","resolution":{"observed_at":"2026-08-02T20:38:35.343597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2602.23058","last_updated":"2026-05-17T06:55:06Z","snapshot_observed_at":"2026-07-06T22:47:11.228912Z","submitted_at":"2026-02-26T14:42:53Z","title":"GeoWorld: Geometric World Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T11:39:15.308355Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2602.23058"},"observation_digest":"sha256:aadbd25a0f5718f4596a541218e382a14c627fc9c5d28cfb04953a070279a020","observation_id":"ed6458fc-c39e-49d3-a453-7e7473b89bd0","resolution":{"observed_at":"2026-05-21T11:40:03.357822Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T20:38:16.567411Z","title":"V-jepa 2: Self- supervised video models enable understanding, prediction and planning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.00167","last_updated":"2026-06-11T07:50:29Z","snapshot_observed_at":"2026-08-02T20:38:15.452636Z","submitted_at":"2026-02-26T09:56:21Z","title":"EgoMoD: Predicting Global Maps of Dynamics from Local Egocentric Observations","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T20:38:16.567411Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.00167"},"observation_digest":"sha256:928134b4ffb6fda02346886553d135d8cab574964ddda58808c8874e0ea30fbc","observation_id":"e85ab659-0887-475e-a06d-6ba5ae6d989d","resolution":{"observed_at":"2026-08-02T20:38:16.567411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T19:58:12.904007Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and plan- ning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.00461","last_updated":"2026-06-10T06:57:00Z","snapshot_observed_at":"2026-08-05T02:58:46.928689Z","submitted_at":"2026-02-28T04:42:34Z","title":"ReMoT: Reinforcement Learning with Motion Contrast Triplets","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T19:58:12.904007Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.00461"},"observation_digest":"sha256:3f50b743446c2ec876529864c00831356c2833a1782013fe73bc78944ff05bfd","observation_id":"1bd60c73-fe88-451c-88f1-86c1b145c81f","resolution":{"observed_at":"2026-08-02T19:58:12.904007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T18:56:44.652978Z","title":"V-jepa 2: Self- supervised video models enable understanding, prediction and plan- ning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.04249","last_updated":"2026-07-07T19:14:10Z","snapshot_observed_at":"2026-08-02T18:56:43.868768Z","submitted_at":"2026-03-04T16:37:40Z","title":"RoboLight: A Dataset with Linearly Composable Illumination for Robotic Manipulation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T18:56:44.652978Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.04249"},"observation_digest":"sha256:523b041d255b2b6cad525e41d1af8fe32bbf9468a306bbba4c7d2aa10ea3aa08","observation_id":"89965aa2-5d0d-493d-8d98-55fb15729978","resolution":{"observed_at":"2026-08-02T18:56:44.652978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2603.07083","last_updated":"2026-04-14T13:41:32Z","snapshot_observed_at":"2026-07-06T22:48:13.053749Z","submitted_at":"2026-03-07T07:41:28Z","title":"Dreamer-CDP: Improving Reconstruction-free World Models Via Continuous Deterministic Representation Prediction","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T14:46:18.645009Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.07083"},"observation_digest":"sha256:b6b0bdcd04b3af8d0c86647cb1bf2bdc5bb2f0e5b28bdb305bea26bc50fdd4f5","observation_id":"06261469-2999-4665-af3d-125c049f3567","resolution":{"observed_at":"2026-05-15T14:50:05.131996Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2603.09030","last_updated":"2026-04-06T01:32:12Z","snapshot_observed_at":"2026-07-30T02:16:48.709523Z","submitted_at":"2026-03-09T23:58:07Z","title":"PlayWorld: Learning Robot World Models from Autonomous Play","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-15T14:05:32.112432Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.09030"},"observation_digest":"sha256:197e476670efb92b5f8bbf88e9b5eda673c5af39dcf2a9b07eaf20d330af1e76","observation_id":"1d328c51-3b8a-4d96-bf3b-8ec7053f88a0","resolution":{"observed_at":"2026-05-15T14:05:54.592109Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T18:11:29.452431Z","title":"V-JEPA 2 : Self-supervised video models enable understanding, prediction and planning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.15553","last_updated":"2026-07-27T17:51:19Z","snapshot_observed_at":"2026-08-04T09:24:38.776155Z","submitted_at":"2026-03-16T17:13:27Z","title":"Self-Distillation of Hidden Layers for Self-Supervised Representation Learning","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-02T18:11:29.452431Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.15553"},"observation_digest":"sha256:f01f564ed5b5a759bb9ed686b6a25eae0b7cd12a679a74c576c431461eb79891","observation_id":"d247fa89-3dd1-4fcb-8121-5ef947af325a","resolution":{"observed_at":"2026-08-02T18:11:29.452431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T17:54:04.043468Z","title":"arXiv preprint arXiv:2506.09985 (2025) 6, 13, 25, 27","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.19235","last_updated":"2026-07-17T06:42:00Z","snapshot_observed_at":"2026-08-02T17:54:03.255042Z","submitted_at":"2026-03-19T17:59:58Z","title":"Generation Models Know Space: Unleashing Implicit 3D Priors for Scene Understanding","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T17:54:04.043468Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.19235"},"observation_digest":"sha256:73b85cb41d06e0b219768780a8863271927fb97d80a75a2e9708bf48d4b9e0c3","observation_id":"16560a48-022d-4352-950b-e8246d764ca0","resolution":{"observed_at":"2026-08-02T17:54:04.043468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2603.19312","last_updated":"2026-06-03T18:50:40Z","snapshot_observed_at":"2026-07-14T21:44:46.992364Z","submitted_at":"2026-03-13T19:48:14Z","title":"LeWorldModel: Stable End-to-End Joint-Embedding Predictive Architecture from Pixels","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T04:09:22.328844Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.19312"},"observation_digest":"sha256:fb6fa497546a79d954157fb3ddf90084ce5835a2332ec788c158eec7c4823e22","observation_id":"23af3160-2c36-400f-84c3-89ff5668f910","resolution":{"observed_at":"2026-05-15T04:09:22.421581Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-07-13T20:17:45.587021Z","title":"arXiv preprint arXiv:2506.09985 (2025) 2, 5, 8, 9, 11","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.22281","last_updated":"2026-06-16T09:39:03Z","snapshot_observed_at":"2026-08-01T16:50:48.503240Z","submitted_at":"2026-03-23T17:59:42Z","title":"ThinkJEPA: Empowering Latent World Models with Large Vision-Language Reasoning Model","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-13T20:17:45.587021Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.22281"},"observation_digest":"sha256:3f9d284c49543178b191486d1af8b7faa961a4ac6273cdabd76555fdd3bb0304","observation_id":"ceb99782-4f30-4b07-9610-178fd1f13b3b","resolution":{"observed_at":"2026-07-13T20:17:45.587021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2603.27134","last_updated":"2026-08-01T21:39:16Z","snapshot_observed_at":"2026-08-05T14:40:51.553162Z","submitted_at":"2026-03-28T04:46:44Z","title":"Factorization Regret mediates compositional generalization in latent space","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-14T23:15:08.716714Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.27134"},"observation_digest":"sha256:685f7547a0162e2daf5955a91b5ad56cd313c87281c597839aa87f5befb5ab22","observation_id":"7f95f2fe-9259-479e-b25b-5e566ce335f8","resolution":{"observed_at":"2026-05-14T23:18:15.909718Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-04T05:46:17.900933Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and planning.ArXiv, abs/2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.27134","last_updated":"2026-08-01T21:39:16Z","snapshot_observed_at":"2026-08-05T14:40:51.553162Z","submitted_at":"2026-03-28T04:46:44Z","title":"Factorization Regret mediates compositional generalization in latent space","version":6},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T05:46:17.900933Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.27134"},"observation_digest":"sha256:7611fe4fb5f9bd576b4580fe8b086f3d857bda3945e494f8ac944728e9386664","observation_id":"100bf105-1541-4c8b-8aa8-2d51137a3a71","resolution":{"observed_at":"2026-08-04T05:46:17.900933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-07-13T15:33:59.448352Z","title":"V-JEPA 2: Self- supervised video models enable understanding, prediction and planning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.29795","last_updated":"2026-05-31T10:37:52Z","snapshot_observed_at":"2026-08-01T18:41:39.635283Z","submitted_at":"2026-03-31T14:28:47Z","title":"Topological sum rule for geometric phases of quantum gates","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-13T15:33:59.448352Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2603.29795"},"observation_digest":"sha256:3cd2c3eafa46909fa68c837c59e4dd692780d848f4ada1faf0df29e2b50f081c","observation_id":"c3bbbd9e-94f2-4f4b-8571-300692b94b8b","resolution":{"observed_at":"2026-07-13T15:33:59.448352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-07-13T14:05:26.303000Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.01985","last_updated":"2026-05-29T16:27:50Z","snapshot_observed_at":"2026-07-13T14:05:19.455452Z","submitted_at":"2026-04-02T12:48:36Z","title":"World Action Verifier: Self-Improving World Models via Forward-Inverse Asymmetry","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-13T14:05:26.303000Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.01985"},"observation_digest":"sha256:db0deae93a198b211913d383b757dafd62d7cd1bede7f21bf9717fab2285acc3","observation_id":"80ecc837-2a2f-4b48-be64-ab3373eb5029","resolution":{"observed_at":"2026-07-13T14:05:26.303000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.03208","last_updated":"2026-06-16T20:46:31Z","snapshot_observed_at":"2026-07-13T13:29:04.173053Z","submitted_at":"2026-04-03T17:32:36Z","title":"Hierarchical Planning with Latent World Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-13T20:13:58.991298Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.03208"},"observation_digest":"sha256:9355abd1cef9f1e6c9cf18d9b71b8f372b0f7e5ba57e03c029815d63fda04085","observation_id":"d094bea9-e17b-486c-9f17-a1f06f2bd0b3","resolution":{"observed_at":"2026-05-13T20:18:13.819302Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.03266","last_updated":"2026-03-18T20:23:52Z","snapshot_observed_at":"2026-08-03T23:17:32.320246Z","submitted_at":"2026-03-18T20:23:52Z","title":"Emergent Compositional Communication for Latent World Properties","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T08:28:10.971822Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.03266"},"observation_digest":"sha256:b3ff7bd60efb230e9d2d9fc82a9ae15f51d0b298ff7d88b9bee402eb5ce868d0","observation_id":"cf57247f-acab-491e-bf07-88a05c31f1ce","resolution":{"observed_at":"2026-05-15T08:29:52.183924Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.04502","last_updated":"2026-04-06T07:57:52Z","snapshot_observed_at":"2026-07-06T22:53:29.118925Z","submitted_at":"2026-04-06T07:57:52Z","title":"Veo-Act: How Far Can Frontier Video Models Advance Generalizable Robot Manipulation?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T20:10:54.362107Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.04502"},"observation_digest":"sha256:6a9887eb53dc0b030712dd946b63943d0d00dccbb60d6641590f384e88302961","observation_id":"b0e9a11d-3fbc-4e1a-a690-7786c981094e","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T19:36:42.100191Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:63f7f7dc13c5378973145dc1a0f2b48b21b0453a5b111447cd04790696772d5f","observation_id":"7323336e-7f29-4016-b2f9-33980d4733dd","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-07-13T09:42:23.808691Z","title":"V-jepa 2: Self-supervised video models enable understanding, prediction and planning.arXiv preprint arXiv:2506.09985, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.04707","last_updated":"2026-05-25T06:28:44Z","snapshot_observed_at":"2026-07-13T09:42:22.607962Z","submitted_at":"2026-04-06T14:19:48Z","title":"OpenWorldLib: A Unified Codebase and Definition of Advanced World Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-13T09:42:23.808691Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.04707"},"observation_digest":"sha256:3affb11608b9220c754f9d78ba68e69841791209e4f6bd779058f34b4ceb14b7","observation_id":"ef4339cd-0695-4a61-8b2a-e82da2a66563","resolution":{"observed_at":"2026-07-13T09:42:23.808691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.05014","last_updated":"2026-04-06T17:59:21Z","snapshot_observed_at":"2026-07-06T22:53:51.418227Z","submitted_at":"2026-04-06T17:59:21Z","title":"StarVLA: A Lego-like Codebase for Vision-Language-Action Model Developing","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T19:01:02.641396Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.05014"},"observation_digest":"sha256:95c7a5381d8e3e91833735028bee69205fadb8a0cd485787f3411419d36019c3","observation_id":"6169bb95-c826-4c1f-af34-866a793f05be","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.06168","last_updated":"2026-04-15T01:21:52Z","snapshot_observed_at":"2026-07-06T22:54:44.656835Z","submitted_at":"2026-04-07T17:59:30Z","title":"Action Images: End-to-End Policy Learning via Multiview Video Generation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T18:51:05.206602Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.06168"},"observation_digest":"sha256:98cf816ee165babce4dc5fc67c2a766a11041d9ad7e7d8088cae9822fd60e3bf","observation_id":"8d1c0a46-fbe9-4a36-9baf-565ffe1c7911","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.07745","last_updated":"2026-04-09T03:03:06Z","snapshot_observed_at":"2026-08-02T14:47:37.509813Z","submitted_at":"2026-04-09T03:03:06Z","title":"The Cartesian Cut in Agentic AI","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T17:54:45.333038Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.07745"},"observation_digest":"sha256:395b90934974cb8fd3e7e252afad1068ffc7232ea6e81a0e8cbd0d7a393d2677","observation_id":"2e3c5bc1-0b32-489a-bc99-9233bafaa622","resolution":{"observed_at":"2026-05-11T05:51:10.828754Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.08460","last_updated":"2026-04-09T16:56:37Z","snapshot_observed_at":"2026-07-06T22:57:31.046146Z","submitted_at":"2026-04-09T16:56:37Z","title":"A Machine Learning Framework for Turbofan Health Estimation via Inverse Problem Formulation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T18:26:48.918721Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.08460"},"observation_digest":"sha256:36134748548625f4a906cdb3c9d26f8a6994b0ee387d13d8cbc353bbab4be06b","observation_id":"c4dbef46-0207-4e20-b674-434e937dacc5","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.08503","last_updated":"2026-05-18T19:22:00Z","snapshot_observed_at":"2026-07-06T22:57:35.435713Z","submitted_at":"2026-04-09T17:48:46Z","title":"Phantom: Physics-Infused Video Generation via Joint Modeling of Visual and Latent Physical Dynamics","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T18:15:37.338442Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.08503"},"observation_digest":"sha256:df1d22bd4f0b2bafeb47abd9fb9cfdd054415b1bb1f284b527804581398aa45c","observation_id":"ff9a67a8-604d-4515-ae69-06bea3cbb133","resolution":{"observed_at":"2026-05-11T05:10:55.621528Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.08503","last_updated":"2026-05-18T19:22:00Z","snapshot_observed_at":"2026-07-06T22:57:35.435713Z","submitted_at":"2026-04-09T17:48:46Z","title":"Phantom: Physics-Infused Video Generation via Joint Modeling of Visual and Latent Physical Dynamics","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T09:17:53.731701Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.08503"},"observation_digest":"sha256:079b3fc3e5096fe673524683bfb5c62a12461323df3e8022453a9e034aed6368","observation_id":"f334355b-f8ce-4e64-914c-c9c4e2cb2118","resolution":{"observed_at":"2026-05-21T09:19:56.619333Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.08995","last_updated":"2026-04-13T03:48:38Z","snapshot_observed_at":"2026-07-06T22:57:59.555972Z","submitted_at":"2026-04-10T06:00:09Z","title":"Matrix-Game 3.0: Real-Time and Streaming Interactive World Model with Long-Horizon Memory","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T18:24:14.349469Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.08995"},"observation_digest":"sha256:8d67bb1f60d85147adc47eaeed05275b78cac1f5aaf11d9ccebe2de9a3c1940c","observation_id":"f6276d56-bbf8-4642-bfac-e566e0fd01c8","resolution":{"observed_at":"2026-05-11T00:41:04.710352Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.10333","last_updated":"2026-04-11T19:32:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-11T19:32:33Z","title":"Zero-shot World Models Are Developmentally Efficient Learners","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T15:33:39.342672Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.10333"},"observation_digest":"sha256:09f1fb2f04dd81bf172190b8c06263ef4e82869dd0a852f0f22190dfb35a9be7","observation_id":"9b339fed-d2bf-4fa8-b583-4f83f8c6f0b8","resolution":{"observed_at":"2026-05-11T10:16:08.390658Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.10385","last_updated":"2026-07-14T05:51:48Z","snapshot_observed_at":"2026-08-02T10:57:31.860855Z","submitted_at":"2026-04-12T00:01:51Z","title":"GTASA: Ground Truth Annotations for Spatiotemporal Analysis, Evaluation and Training of Video Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T16:38:10.956726Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.10385"},"observation_digest":"sha256:5a3420698efbf07a7d54cfc7a0a56e77d2a2029c1b664d16a0a45ed65421bf77","observation_id":"9b94b971-c780-4c1d-a3f7-406da2b4b39a","resolution":{"observed_at":"2026-05-11T08:26:01.705672Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.11415","last_updated":"2026-04-13T13:00:29Z","snapshot_observed_at":"2026-07-06T22:59:49.614780Z","submitted_at":"2026-04-13T13:00:29Z","title":"Observe Less, Understand More: Cost-aware Cross-scale Observation for Remote Sensing Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T15:18:11.837448Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.11415"},"observation_digest":"sha256:ab7a6b0a9b098758395cf1ae33c3f0d4dfe455a6b6e1976e8068d664da3611cb","observation_id":"50996131-99c6-4bb5-8670-9375b8c28b1c","resolution":{"observed_at":"2026-05-11T10:51:02.844753Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.11707","last_updated":"2026-04-13T16:42:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-13T16:42:46Z","title":"Representations Before Pixels: Semantics-Guided Hierarchical Video Prediction","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:53.578491Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.11707"},"observation_digest":"sha256:2ff2cc8e3a5932b05831162a61f87dbcee196205becd3e5a3461511700780233","observation_id":"d5400338-cc49-433f-bb25-0d875da0e26f","resolution":{"observed_at":"2026-05-11T08:45:58.539499Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.11751","last_updated":"2026-04-13T17:25:41Z","snapshot_observed_at":"2026-07-06T23:00:03.762376Z","submitted_at":"2026-04-13T17:25:41Z","title":"Grounded World Model for Semantically Generalizable Planning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T15:05:29.465402Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.11751"},"observation_digest":"sha256:f2960afd5132f40edb14471939959e990fcd159d4bdcd302cc0d2c8b06de52e2","observation_id":"e0490e0e-8127-4f85-9983-874e5fea1123","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.13015","last_updated":"2026-04-27T15:02:49Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:54:17Z","title":"Learning Versatile Humanoid Manipulation with Touch Dreaming","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T14:48:33.942119Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.13015"},"observation_digest":"sha256:fdd37e68a9a5311290e059df98dcc87f40fd56f74ab7ba3503785793e99ba2ca","observation_id":"57faeb5d-c3c7-4e0a-a986-b4f43b48116d","resolution":{"observed_at":"2026-05-11T11:31:03.269322Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.14816","last_updated":"2026-04-16T09:36:58Z","snapshot_observed_at":"2026-08-03T04:36:59.454952Z","submitted_at":"2026-04-16T09:36:58Z","title":"NTIRE 2026 Challenge on Video Saliency Prediction: Methods and Results","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T12:23:07.074182Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.14816"},"observation_digest":"sha256:a1d07f03879008c7370a2a1f8b6afa13087c88967ff744e21df78b01cb78773e","observation_id":"bc09f644-da8c-42d7-86ed-5f21a588671c","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.15299","last_updated":"2026-04-16T17:57:08Z","snapshot_observed_at":"2026-07-06T23:02:53.677687Z","submitted_at":"2026-04-16T17:57:08Z","title":"AnimationBench: Are Video Models Good at Character-Centric Animation?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T11:47:27.131584Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.15299"},"observation_digest":"sha256:98c964fb20d422b5abdb063baa753127af2d69c9574f64c90893c36b43fcc2b3","observation_id":"c071330a-ffda-485a-9964-3a03dad4a69f","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.16086","last_updated":"2026-04-17T14:15:12Z","snapshot_observed_at":"2026-07-06T23:03:30.601598Z","submitted_at":"2026-04-17T14:15:12Z","title":"Stylistic-STORM (ST-STORM) : Perceiving the Semantic Nature of Appearance","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T08:59:40.341143Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.16086"},"observation_digest":"sha256:ae8e6b815a48e687806e06d769e3daa4ec02c45991f8ac968cec8584d9204056","observation_id":"4a1bb3cd-6d4e-48ba-813c-d47acc0a6cd3","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.16592","last_updated":"2026-04-17T17:51:46Z","snapshot_observed_at":"2026-07-06T23:03:52.631613Z","submitted_at":"2026-04-17T17:51:46Z","title":"Human Cognition in Machines: A Unified Perspective of World Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T08:12:15.663761Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.16592"},"observation_digest":"sha256:c25952b435d8872c4069dfa8d11927b61dd36fb141b088708c4f996f010a6f40","observation_id":"aa9b2aa5-4bde-4283-888b-d13bc903e180","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.16733","last_updated":"2026-04-17T22:43:49Z","snapshot_observed_at":"2026-07-06T23:04:01.558812Z","submitted_at":"2026-04-17T22:43:49Z","title":"Active World-Model with 4D-informed Retrieval for Exploration and Awareness","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T08:20:40.569482Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.16733"},"observation_digest":"sha256:06dc7723b2e2df35253c0a6df1d3e6ebf182509a4cbbe5313e2e690fd0009219","observation_id":"f054379c-ff58-47a3-ab08-0affcc7c8d33","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.16843","last_updated":"2026-04-18T05:25:18Z","snapshot_observed_at":"2026-07-06T23:04:05.813982Z","submitted_at":"2026-04-18T05:25:18Z","title":"Watching Physics: the Generative Science of Matter and Motion","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T07:32:27.962931Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.16843"},"observation_digest":"sha256:6c2214f418c963e4dad7b8d360dc5f9da4f41e1d0976788c0c081f16a309c537","observation_id":"98c68955-945f-49bb-9590-6929e3521b14","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.19092","last_updated":"2026-05-14T07:32:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-21T05:09:56Z","title":"RoboWM-Bench: A Benchmark for Evaluating World Models in Robotic Manipulation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T03:02:26.084185Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.19092"},"observation_digest":"sha256:b588542f2d21786680c173031f7a8333099df809cf4e3940b8f7f86c9d9d951c","observation_id":"7b640478-bbcf-40c1-9021-ef2fd37152b5","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.19092","last_updated":"2026-05-14T07:32:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-21T05:09:56Z","title":"RoboWM-Bench: A Benchmark for Evaluating World Models in Robotic Manipulation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T06:15:32.881140Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.19092"},"observation_digest":"sha256:f1a8072b875275b458aac8a4534f000eafecfcc8c28aa247d834d339b3d0ef74","observation_id":"b8663770-f9a9-4c91-b695-9640e57bf87d","resolution":{"observed_at":"2026-05-15T06:19:50.267503Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.19683","last_updated":"2026-04-22T17:44:56Z","snapshot_observed_at":"2026-07-06T23:06:16.972438Z","submitted_at":"2026-04-21T17:05:37Z","title":"Mask World Model: Predicting What Matters for Robust Robot Policy Learning","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T02:14:17.676675Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.19683"},"observation_digest":"sha256:10b374b147dc8e05b6d8131745a0cba1873a484d24de3041eb1acebbef75f6e2","observation_id":"3a9e2724-55be-464a-840f-c73f6113e49c","resolution":{"observed_at":"2026-05-11T13:11:06.331616Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.20246","last_updated":"2026-04-22T06:49:12Z","snapshot_observed_at":"2026-07-06T23:06:44.624357Z","submitted_at":"2026-04-22T06:49:12Z","title":"Cortex 2.0: Grounding World Models in Real-World Industrial Deployment","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T00:52:05.348704Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.20246"},"observation_digest":"sha256:b53d0df55cb73da8ee672609a0dae3cbb08a596c9616c9d8da7ef1e1ef218194","observation_id":"2f9ddda3-64b1-4c17-a97c-023df2cd2804","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.20760","last_updated":"2026-04-22T16:48:43Z","snapshot_observed_at":"2026-08-04T10:29:14.353844Z","submitted_at":"2026-04-22T16:48:43Z","title":"Exploring High-Order Self-Similarity for Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T00:59:03.890135Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.20760"},"observation_digest":"sha256:1a3980807757e7b8cfbfe19bcc1ccca7a0f6408f4a63047c847c8c76c8c21d2d","observation_id":"70f8e283-5974-4e01-9be8-c535914a03ee","resolution":{"observed_at":"2026-05-11T00:33:51.269344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.21017","last_updated":"2026-06-04T17:26:29Z","snapshot_observed_at":"2026-08-03T16:24:19.318353Z","submitted_at":"2026-04-22T19:05:17Z","title":"Open-H-Embodiment: A Large-Scale Dataset for Enabling Foundation Models in Medical Robotics","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-09T23:53:09.614994Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.21017"},"observation_digest":"sha256:162b986942b620434d286f06ef6cf513837349d2f6d877056225562e75b40a23","observation_id":"c8f847a4-e859-4099-9380-7d3e2a2d9cf7","resolution":{"observed_at":"2026-05-11T13:56:05.568412Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.21681","last_updated":"2026-04-23T13:45:32Z","snapshot_observed_at":"2026-08-01T17:21:19.902700Z","submitted_at":"2026-04-23T13:45:32Z","title":"Sapiens2","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-09T21:59:43.755956Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.21681"},"observation_digest":"sha256:156eb06458dab7670dd1356d066df4f02a36992aa0c4e66a3039a7e22bb0a607","observation_id":"6e70a9cc-382c-40c9-8444-cc4252b97c92","resolution":{"observed_at":"2026-05-11T14:21:06.834481Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.21780","last_updated":"2026-04-23T15:38:11Z","snapshot_observed_at":"2026-07-06T23:08:19.446684Z","submitted_at":"2026-04-23T15:38:11Z","title":"Only Brains Align with Brains: Cross-Region Alignment Patterns Expose Limits of Normative Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-08T13:13:03.129195Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.21780"},"observation_digest":"sha256:ce15e3a2cb6979036cb3fd2e4d47f42c8822ba66fd3fe62fc3570adbf421cc36","observation_id":"32102979-e1a1-4f54-b2ea-65e91e5a982a","resolution":{"observed_at":"2026-05-11T18:56:07.566890Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.22227","last_updated":"2026-04-29T07:51:58Z","snapshot_observed_at":"2026-08-03T03:59:21.113148Z","submitted_at":"2026-04-24T05:02:20Z","title":"A Co-Evolutionary Theory of Human-AI Coexistence: Mutualism, Governance, and Dynamics in Complex Societies","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-08T09:51:33.783535Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.22227"},"observation_digest":"sha256:1b8b678219c3f78d4d3c87d4680aef552a56dc3bc296cd0d58a4b7492010807b","observation_id":"33747b10-1494-4a32-a753-d5d9f3fc89ac","resolution":{"observed_at":"2026-05-11T20:16:08.977561Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T12:32:40.901242Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:8598d0bf56acdeea81f1fc0e4169e60ada207f148cb5fcfad354016ff0ff56ba","observation_id":"bd7e36e3-8c87-49ea-940a-dfe051c0c54e","resolution":{"observed_at":"2026-05-11T19:06:12.301942Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T01:57:09.197785Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:9d690eeb5eab632ed4784fb3c07a43d974747b8d8a5d360c429b4ad8790f382d","observation_id":"5dc7b409-eda9-4c7f-9e76-954eb13f19f5","resolution":{"observed_at":"2026-05-12T07:46:30.201939Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.22686","last_updated":"2026-05-13T16:46:06Z","snapshot_observed_at":"2026-07-06T23:09:01.221922Z","submitted_at":"2026-04-24T16:12:44Z","title":"SS3D: End2End Self-Supervised 3D from Web Videos","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-14T21:26:38.410299Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.22686"},"observation_digest":"sha256:3ec1ed4181738f0f97b888001fd071aced70669edf4670f0758528f000bde28f","observation_id":"a7c581a3-6796-411e-ba2a-f53831bca7df","resolution":{"observed_at":"2026-05-14T21:27:59.609043Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.22748","last_updated":"2026-06-08T04:24:27Z","snapshot_observed_at":"2026-08-02T10:58:24.119885Z","submitted_at":"2026-04-24T17:48:47Z","title":"Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-08T12:02:07.027775Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.22748"},"observation_digest":"sha256:402806e3f820acd6f19dae05d4c7ac5ea33d41f1a9ca09f8ddbe573abb0b339f","observation_id":"bfc17488-b348-4be3-9279-508eedc1a782","resolution":{"observed_at":"2026-05-11T19:26:07.829429Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.22748","last_updated":"2026-06-08T04:24:27Z","snapshot_observed_at":"2026-08-02T10:58:24.119885Z","submitted_at":"2026-04-24T17:48:47Z","title":"Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-07-04T17:29:43.764085Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.22748"},"observation_digest":"sha256:4770f5de0bb4e9f7ad471972eab3a7106100eaa09225de46daca20cb9f386e15","observation_id":"3483b91e-7ff2-4218-b80f-a014b115de23","resolution":{"observed_at":"2026-07-04T17:29:59.538944Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.24681","last_updated":"2026-05-21T05:08:13Z","snapshot_observed_at":"2026-08-02T19:33:59.938088Z","submitted_at":"2026-04-27T16:42:18Z","title":"Learning Human-Intention Priors from Large-Scale Human Demonstrations for Robotic Manipulation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-08T02:51:27.662262Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.24681"},"observation_digest":"sha256:dce686cc420c031e0f0bff3ddb3c48f07420838f103c3b11262639e4987ac73e","observation_id":"77f8cd15-cd30-45c4-8b22-2ecc5e7e580f","resolution":{"observed_at":"2026-05-11T22:26:11.646080Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.24681","last_updated":"2026-05-21T05:08:13Z","snapshot_observed_at":"2026-08-02T19:33:59.938088Z","submitted_at":"2026-04-27T16:42:18Z","title":"Learning Human-Intention Priors from Large-Scale Human Demonstrations for Robotic Manipulation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-22T11:16:58.104663Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.24681"},"observation_digest":"sha256:5199edbff63531a34f6785f7e096102e7baf0c775a29303995b0c1aee8115201","observation_id":"59f04218-6cf5-4c70-bc8d-1b8089e884b0","resolution":{"observed_at":"2026-05-22T11:21:29.203425Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":"2506.09985","doi":"10.48550/arxiv.2506.09985","metadata_source":"pith","pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","venue":"cs.AI","work_id":"a9c28401-f16a-4933-89f0-788e2f94e52b","year":2025},"citing_paper":{"arxiv_id":"2604.26182","last_updated":"2026-07-21T05:55:59Z","snapshot_observed_at":"2026-08-02T17:21:10.469702Z","submitted_at":"2026-04-28T23:59:19Z","title":"Lifting Embodied World Models for Planning and Control","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-07T16:32:46.059831Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.26182"},"observation_digest":"sha256:74f558f16c0c985d2b6319ac1c9bee2375441c8a08939cfc63438a90761e1d4e","observation_id":"7198b9f8-18c0-452e-843c-25599a2684a7","resolution":{"observed_at":"2026-05-11T23:41:15.767575Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:37.559938+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.09985","snapshot_observed_at":"2026-08-02T15:20:24.269257Z","title":"arXiv preprint arXiv:2506.09985 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.26182","last_updated":"2026-07-21T05:55:59Z","snapshot_observed_at":"2026-08-02T17:21:10.469702Z","submitted_at":"2026-04-28T23:59:19Z","title":"Lifting Embodied World Models for Planning and Control","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T15:20:24.269257Z"},"links":{"cited_paper":"/paper/2506.09985","citing_paper":"/paper/2604.26182"},"observation_digest":"sha256:5f7a597d6b6765542faf8881cc50f6a280de7d0124051918d2e4811d0c36adc9","observation_id":"bd2d379d-f512-49d7-8a73-019fc8909cf0","resolution":{"observed_at":"2026-08-02T15:20:24.269257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.09985/citation-record","integrity":"/paper/2506.09985/integrity","json":"/paper/2506.09985/citation-record.json","paper":"/paper/2506.09985"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.03575","last_updated":"2025-07-09T19:35:31Z","snapshot_observed_at":"2026-08-03T00:21:10.886100Z","submitted_at":"2025-01-07T06:55:50Z","title":"Cosmos World Foundation Model Platform for Physical AI","version":3},"cited_work":{"arxiv_id":"2501.03575","doi":"10.48550/arxiv.2501.03575","metadata_source":"pith","pith_arxiv_id":"2501.03575","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Cosmos World Foundation Model Platform for Physical AI","venue":"cs.CV","work_id":"a2dba24c-318d-476a-8b21-4289c265810c","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2501.03575","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:1fe5fa619906397ee2b043321b67f7d1f25fe85ee5b20ca90939637b8536e7ce","observation_id":"201bc43f-a55d-4a40-abd1-5a4ab1bd16a4","resolution":{"observed_at":"2026-05-11T00:33:50.706975Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-11T15:53:29.560535+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T15:53:29.560535+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.07277","last_updated":"2022-10-13T18:10:01Z","snapshot_observed_at":"2026-07-06T14:04:55.099373Z","submitted_at":"2022-10-13T18:10:01Z","title":"The Hidden Uniform Cluster Prior in Self-Supervised Learning","version":1},"cited_work":{"arxiv_id":"2210.07277","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2210.07277","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2210.07277 , year=","venue":null,"work_id":"9a6adc1a-b202-446a-9c38-99a37b840613","year":null},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2210.07277","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:69d3e870f370266edb84b7ae7775f5e0ed19be7b9e9267dbb3ac1d1d6e58ad6c","observation_id":"84124e4f-7ec8-46d3-9e7c-2f185788a459","resolution":{"observed_at":"2026-05-11T00:33:50.615165Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08471","last_updated":"2024-02-15T18:59:11Z","snapshot_observed_at":"2026-08-02T05:40:31.086014Z","submitted_at":"2024-02-15T18:59:11Z","title":"Revisiting Feature Prediction for Learning Visual Representations from Video","version":1},"cited_work":{"arxiv_id":"2404.08471","doi":"10.1145/3178876.3185996","metadata_source":"pith","pith_arxiv_id":"2404.08471","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Revisiting Feature Prediction for Learning Visual Representations from Video","venue":"cs.CV","work_id":"f7251dcf-5341-4915-bfe7-27812387b61a","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2404.08471","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:0ddc7636b76c192221e0a102b9216e58cedf8817c42cb5660f2361c1d4f517d0","observation_id":"0c4c3889-4ac4-43e6-b0b7-542a2be6494a","resolution":{"observed_at":"2026-05-12T12:40:24.452580Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-23T01:54:34.489718+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T01:54:34.489718+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14734","last_updated":"2025-03-27T02:52:43Z","snapshot_observed_at":"2026-08-02T04:15:31.100670Z","submitted_at":"2025-03-18T21:06:21Z","title":"GR00T N1: An Open Foundation Model for Generalist Humanoid Robots","version":2},"cited_work":{"arxiv_id":"2503.14734","doi":"10.48550/arxiv.2503.14734","metadata_source":"pith","pith_arxiv_id":"2503.14734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GR00T N1: An Open Foundation Model for Generalist Humanoid Robots","venue":"cs.RO","work_id":"e2db69c7-ee8a-4cb7-a761-7b8de1dfcf97","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2503.14734","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:3b55615d18977244f0f65c2229f049de190c708bfd547127bd3a09309c6ec63f","observation_id":"1cac71f1-58df-4132-bd8c-e6bfe8e9cf69","resolution":{"observed_at":"2026-05-11T00:33:50.658349Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-12T08:49:11.206777+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T08:49:11.206777+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.24164","last_updated":"2026-01-08T17:01:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-31T17:22:30Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","version":4},"cited_work":{"arxiv_id":"2410.24164","doi":"10.48550/arxiv.2410.24164","metadata_source":"pith","pith_arxiv_id":"2410.24164","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","venue":"cs.LG","work_id":"f790abdc-a796-482f-a40d-f8ee035ecfc2","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2410.24164","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:266cf1c64064d9fee2da5c67b7baacb9d6396c2d8d08ad798c814135ab0ee12e","observation_id":"5b6f6463-1cb4-4704-8729-e5b45bc57d45","resolution":{"observed_at":"2026-05-11T00:33:50.663856Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16054","last_updated":"2025-04-22T17:31:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:31:29Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","version":1},"cited_work":{"arxiv_id":"2504.16054","doi":"10.1609/aaai.v40i28.39562","metadata_source":"pith","pith_arxiv_id":"2504.16054","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","venue":"cs.LG","work_id":"d1ad7304-d09a-49bc-809e-846439f6aff9","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2504.16054","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:b69bb721d7b2ef7432b0f5503ca431625526d18bcb27018d20f5a3f3b07af5c3","observation_id":"72dc4682-8985-4280-81fd-d359e9130445","resolution":{"observed_at":"2026-05-11T00:33:50.677015Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13181","last_updated":"2025-04-28T18:01:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-17T17:59:57Z","title":"Perception Encoder: The best visual embeddings are not at the output of the network","version":2},"cited_work":{"arxiv_id":"2504.13181","doi":"10.48550/arxiv.2504.13181","metadata_source":"pith","pith_arxiv_id":"2504.13181","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Perception Encoder: The best visual embeddings are not at the output of the network","venue":"cs.CV","work_id":"409be941-4d4a-4ceb-a28a-eaa2d7709a1c","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2504.13181","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:827957a966a1fdd42cb7057869703286ed2e8500354c56df3b138410af7a8d03","observation_id":"a641c187-947a-4c2a-a8d5-7f5e6f296f4c","resolution":{"observed_at":"2026-05-13T22:21:16.084670Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":"2307.15818","doi":"10.48550/arxiv.2307.15818","metadata_source":"pith","pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","venue":"cs.RO","work_id":"ff438a8a-8003-4fae-9131-acd418b3597b","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:cad3a1d17f1bc3053f181e249458a40df817f4774c670d032885e341c674cc58","observation_id":"c6b04c45-8f01-4280-97ac-3488f0dd72d2","resolution":{"observed_at":"2026-05-11T00:33:50.699817Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10818","last_updated":"2024-10-15T17:55:46Z","snapshot_observed_at":"2026-07-06T19:33:16.949210Z","submitted_at":"2024-10-14T17:59:58Z","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","version":2},"cited_work":{"arxiv_id":"2410.10818","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.10818","snapshot_observed_at":"2026-07-03T23:39:03.234328Z","title":"Temporalbench: Benchmarking fine-grained temporal understanding for multimodal video models","venue":null,"work_id":"0f787776-4151-41f7-8d50-6174eb1340d3","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2410.10818","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:f35a8837bbaa3c870a9cc524ae35e5eb75c758cf1d8ff51a4ff5eceba676a0ed","observation_id":"b3e7157c-7b90-4817-989b-f5200612b389","resolution":{"observed_at":"2026-05-11T00:33:50.948202Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.01340","last_updated":"2018-08-03T20:17:05Z","snapshot_observed_at":"2026-07-06T06:54:00.483796Z","submitted_at":"2018-08-03T20:17:05Z","title":"A Short Note about Kinetics-600","version":1},"cited_work":{"arxiv_id":"1808.01340","doi":null,"metadata_source":"pith","pith_arxiv_id":"1808.01340","snapshot_observed_at":"2026-07-09T11:16:11.465544Z","title":"A Short Note about Kinetics-600","venue":"cs.CV","work_id":"851b1623-6feb-441e-8849-b07f1753f22e","year":2018},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1808.01340","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:83f8d391895bce76293ddd4f6edf3a77a6b54a5d797610a2f69bef1008d1fae4","observation_id":"44f9af9e-5d1a-4739-96ae-26e3ee04a784","resolution":{"observed_at":"2026-05-11T00:33:50.714339Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.06987","last_updated":"2022-10-17T19:58:40Z","snapshot_observed_at":"2026-07-06T08:08:03.081236Z","submitted_at":"2019-07-15T12:58:21Z","title":"A Short Note on the Kinetics-700 Human Action Dataset","version":2},"cited_work":{"arxiv_id":"1907.06987","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1907.06987","snapshot_observed_at":"2026-07-04T09:49:44.399575Z","title":"A short note on the kinetics- 700 human action dataset","venue":null,"work_id":"6eaebe96-5819-4220-84d5-dd888c79b8f4","year":1907},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1907.06987","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:51aa877af41933c337b84e0a3da1e7a5ff0ec4fd05a611de5bc1be2e3c8b4a1f","observation_id":"cb071916-8124-410a-98cb-d137010fa565","resolution":{"observed_at":"2026-05-11T00:33:50.725871Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15212","last_updated":"2025-07-09T16:58:07Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:59:51Z","title":"Scaling 4D Representations","version":2},"cited_work":{"arxiv_id":"2412.15212","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.15212","snapshot_observed_at":"2026-07-10T00:26:39.628379Z","title":"Scaling 4d representations","venue":"cs.CV","work_id":"d708ea1a-01d4-401d-ada6-b36a346e7bf9","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2412.15212","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:6b7e2dc4d17d29631de84da7bf92475401c3044900dbccc7cb2fc1408aec5e3d","observation_id":"3f95764d-646c-414b-9d2c-358331e7599a","resolution":{"observed_at":"2026-05-11T00:33:50.730654Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.07749","last_updated":"2021-06-10T23:54:34Z","snapshot_observed_at":"2026-07-06T11:00:04.987477Z","submitted_at":"2021-04-15T20:10:11Z","title":"Actionable Models: Unsupervised Offline Reinforcement Learning of Robotic Skills","version":3},"cited_work":{"arxiv_id":"2104.07749","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2104.07749","snapshot_observed_at":"2026-06-29T06:53:12.435264Z","title":"arXiv preprint arXiv:2104.07749 , year=","venue":null,"work_id":"9f8cacca-a18f-412d-9656-7a5274a3531c","year":2021},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2104.07749","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:f471539f4b10a9e66b01bb1e38d8a14fb9da757545830e0ba4f39d509d58850a","observation_id":"cb939337-9d3a-4bff-99dd-4dab5c33a910","resolution":{"observed_at":"2026-05-11T00:33:50.741002Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":"2412.05271","doi":"10.48550/arxiv.2412.05271","metadata_source":"pith","pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","venue":"cs.CV","work_id":"ee70bdc8-4656-4849-ada7-ce42a2278d70","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:84272e84ffc6ef6466d688c5105e97d849f8891c522dd46c46c573d39b1f1dab","observation_id":"3cd5d2a7-0d85-490e-a597-fadc8b1949d0","resolution":{"observed_at":"2026-05-11T00:33:50.745932Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13180","last_updated":"2025-07-23T19:22:35Z","snapshot_observed_at":"2026-08-01T20:25:06.490113Z","submitted_at":"2025-04-17T17:59:56Z","title":"PerceptionLM: Open-Access Data and Models for Detailed Visual Understanding","version":3},"cited_work":{"arxiv_id":"2504.13180","doi":"10.48550/arxiv.2504.13180","metadata_source":"arxiv_reference","pith_arxiv_id":"2504.13180","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Perceptionlm: Open-access data and models for detailed visual understanding","venue":"ArXiv.org","work_id":"e81d25ef-bd13-473f-8fc5-0cf3dc3af528","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2504.13180","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:91de967dec5b23b1b44f485e70b6273cf9636ca4164e41263494d8ae09e9f89e","observation_id":"5481e072-022d-4fb6-974c-a1f73ccbf903","resolution":{"observed_at":"2026-05-11T00:33:50.751197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07752","last_updated":"2025-03-25T09:46:02Z","snapshot_observed_at":"2026-08-05T14:33:12.627325Z","submitted_at":"2024-10-10T09:28:36Z","title":"Lost in Time: A New Temporal Benchmark for VideoLLMs","version":3},"cited_work":{"arxiv_id":"2410.07752","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.07752","snapshot_observed_at":"2026-07-02T19:07:17.473192Z","title":"Lost in time: A new temporal benchmark for videollms","venue":null,"work_id":"8eb5b83d-8383-4120-a32e-cfa7e77ae46a","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2410.07752","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:26b12a742354c0da5d624db6814c6e8e04453397a30e231e694ea51f37896453","observation_id":"2ddcf280-f912-4b04-a12e-5e84b7c485e2","resolution":{"observed_at":"2026-05-11T00:33:50.755434Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s11263-021-01531-2","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rescaling egocentric vision: Collection, pipeline and challenges for epic-kitchens-100.International Journal of Computer Vision (IJCV)","venue":"International Journal of Computer Vision","work_id":"8b03dda4-6221-468a-b196-4dde64c9b5b7","year":2022},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:16be632426408dfaea6658a56060ea0e1f3fd5fabca74a4b00e65fb9cdc2f046","observation_id":"1605c1b9-48e0-4470-8876-c126e7c7da52","resolution":{"observed_at":"2026-05-11T00:33:50.598343Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-20T11:23:28.974089+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T11:23:28.974089+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":"2010.11929","doi":"10.1175/jcli-d-22-0357.1","metadata_source":"pith","pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","venue":"cs.CV","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","year":2020},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:2314788362aa49cd30c5f44d5d0c975e1d89145fd6d02d3cda16451db0c2ae9f","observation_id":"26548820-34cc-46ec-b905-9e3d7a804b1f","resolution":{"observed_at":"2026-05-11T00:33:50.759248Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2023.03378","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Driess, F","venue":null,"work_id":"e0fc7630-e8e5-416f-9f96-df9f5236ece9","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:8f94ca580d5e53265ad8de7136b0697f1418b9d69e09a9f8b1dffee2ee835627","observation_id":"57437d9d-66ca-4e5a-87c3-adf9bf9e453e","resolution":{"observed_at":"2026-05-11T00:33:50.763515Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.00568","last_updated":"2018-12-03T06:06:25Z","snapshot_observed_at":"2026-07-06T07:18:36.356308Z","submitted_at":"2018-12-03T06:06:25Z","title":"Visual Foresight: Model-Based Deep Reinforcement Learning for Vision-Based Robotic Control","version":1},"cited_work":{"arxiv_id":"1812.00568","doi":null,"metadata_source":"pith","pith_arxiv_id":"1812.00568","snapshot_observed_at":"2026-07-04T06:49:37.964200Z","title":"Visual Foresight: Model-Based Deep Reinforcement Learning for Vision-Based Robotic Control","venue":"cs.RO","work_id":"4e0d30bd-36e6-464f-84b3-1e5f4f95cb7d","year":2018},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1812.00568","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:051e192bb16e79dcdc19d645e0c8183555a0c11b9f95ce1bea6daf9d4ce8a43b","observation_id":"fcb8aa3a-2c89-4015-82af-6b32beb1c621","resolution":{"observed_at":"2026-05-11T00:33:50.779946Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01017","last_updated":"2025-04-01T17:59:15Z","snapshot_observed_at":"2026-08-03T20:04:29.599244Z","submitted_at":"2025-04-01T17:59:15Z","title":"Scaling Language-Free Visual Representation Learning","version":1},"cited_work":{"arxiv_id":"2504.01017","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.01017","snapshot_observed_at":"2026-07-04T19:50:09.839784Z","title":"Scaling language-free visual representation learning.arXiv preprint arXiv:2504.01017","venue":null,"work_id":"99aa7a09-2b4a-446c-b623-3f34329c0d46","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2504.01017","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:fb2c9145ad0758bd6660547e6949a355db02b3b29642050215c0268172664849","observation_id":"0b4fc876-9190-4d61-8583-94c60ded441c","resolution":{"observed_at":"2026-05-11T00:33:50.790701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14402","last_updated":"2024-11-21T18:31:25Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:31:25Z","title":"Multimodal Autoregressive Pre-training of Large Vision Encoders","version":1},"cited_work":{"arxiv_id":"2411.14402","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14402","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Multimodal autoregressive pre-training of large vision encoders","venue":null,"work_id":"1982a4f1-7cdd-448e-8ddd-85cd782e6e17","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2411.14402","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:9b98c1850cab15f705b5ed759af3d4514a42d5ac0e47fdfa549749b7e80ffa48","observation_id":"0d1cdbad-61e0-4aa1-8757-c176ebea21e6","resolution":{"observed_at":"2026-05-11T00:33:50.796334Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.07404","last_updated":"2016-01-19T20:58:24Z","snapshot_observed_at":"2026-07-06T04:37:31.728881Z","submitted_at":"2015-11-23T20:27:48Z","title":"Learning Visual Predictive Models of Physics for Playing Billiards","version":3},"cited_work":{"arxiv_id":"1511.07404","doi":null,"metadata_source":"pith","pith_arxiv_id":"1511.07404","snapshot_observed_at":"2026-07-03T05:27:39.821170Z","title":"Learning Visual Predictive Models of Physics for Playing Billiards","venue":"cs.CV","work_id":"fc566d7a-3e40-4c2c-924a-4971f6de0f50","year":2015},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1511.07404","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:56c3c71da8221eb66e121d70f0741ec07dd8607e60a1170037a4d44cea0dab0b","observation_id":"1df28384-6207-4a77-8bf9-e715e6d7bd4f","resolution":{"observed_at":"2026-05-11T00:33:50.811345Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"records/1260860","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T14:47:14.558044Z","title":"He, B., Yin, L., Zhen, H.-L., Liu, S., Wu, H., Zhang, X., Yuan, M., and Ma, C","venue":null,"work_id":"e7e9d443-273d-4722-8ec7-59adb893de0e","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:84d3689744c2826006526134e951f7779214fb2f043000ced0a091b2e434e41f","observation_id":"da9481c3-0afc-4f49-993a-4ed25268cb13","resolution":{"observed_at":"2026-05-11T00:33:50.821430Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:ed6aedb71f1d8fbe1465b38d36961305b924ff6d4d00b96c3063bb83d32edcf4","observation_id":"3cb7acec-43f4-4ac7-8ef5-9fbac5a594c6","resolution":{"observed_at":"2026-05-11T00:33:50.829441Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.11894","last_updated":"2022-08-06T10:09:47Z","snapshot_observed_at":"2026-07-06T13:24:06.368922Z","submitted_at":"2022-06-23T17:59:33Z","title":"MaskViT: Masked Visual Pre-Training for Video Prediction","version":2},"cited_work":{"arxiv_id":"2206.11894","doi":"10.48550/arxiv.2206.11894","metadata_source":"arxiv_reference","pith_arxiv_id":"2206.11894","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gupta, S","venue":"arXiv (Cornell University)","work_id":"e5efab6e-7f7e-485c-83f5-1f69f77d39d9","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2206.11894","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:2a5b8089094ae0c63c67c091024374840c0a227ea069e6e2eeff7f67a348d614","observation_id":"c7a600d1-840b-4ff7-86ef-153bff29820d","resolution":{"observed_at":"2026-05-11T00:33:50.835310Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.10122","last_updated":"2018-05-09T09:06:27Z","snapshot_observed_at":"2026-07-31T21:36:45.596575Z","submitted_at":"2018-03-27T15:08:55Z","title":"World Models","version":4},"cited_work":{"arxiv_id":"1803.10122","doi":"10.48550/arxiv.1712.00409","metadata_source":"pith","pith_arxiv_id":"1803.10122","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"World Models","venue":"cs.LG","work_id":"07227eee-8445-4c98-bce4-c6a6fd5ed907","year":2018},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1803.10122","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:3cec764e6ecdf803a3d0d36d5a77278abbd6fca4adfe7ca5dc139c917ce0fddd","observation_id":"cbbef07d-29b2-47c4-95d4-fddd0f75e9ac","resolution":{"observed_at":"2026-05-11T03:08:36.665942Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.01603","last_updated":"2020-03-17T17:10:58Z","snapshot_observed_at":"2026-08-03T15:20:23.515607Z","submitted_at":"2019-12-03T18:57:16Z","title":"Dream to Control: Learning Behaviors by Latent Imagination","version":3},"cited_work":{"arxiv_id":"1912.01603","doi":"10.48550/arxiv.1912.01603","metadata_source":"pith","pith_arxiv_id":"1912.01603","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Dream to Control: Learning Behaviors by Latent Imagination","venue":"cs.LG","work_id":"5103f4be-344a-4139-8504-eaa59f5bac9d","year":2019},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1912.01603","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:65cd555154492e116853cc6495d29a41ddd3854a260c600950a7b98e660dd740","observation_id":"19a1361b-43c3-48e1-aff6-8e92026358e7","resolution":{"observed_at":"2026-05-12T01:16:36.618198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.04955","last_updated":"2022-07-19T18:14:36Z","snapshot_observed_at":"2026-07-06T12:46:08.838773Z","submitted_at":"2022-03-09T18:58:28Z","title":"Temporal Difference Learning for Model Predictive Control","version":2},"cited_work":{"arxiv_id":"2203.04955","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.04955","snapshot_observed_at":"2026-07-04T19:00:06.256105Z","title":"Temporal difference learning for model predictive control","venue":null,"work_id":"7bad3438-3b8c-438f-98c5-122e3c23095a","year":2022},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2203.04955","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:fd756909baba1fdea83db00592faf16e4291b6bde56d7f0de1bd3d0de4039336","observation_id":"5754e964-70a6-4abd-a996-3b33939c6594","resolution":{"observed_at":"2026-05-11T00:33:50.866977Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16828","last_updated":"2024-03-21T17:56:19Z","snapshot_observed_at":"2026-07-31T05:32:29.431480Z","submitted_at":"2023-10-25T17:57:07Z","title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","version":2},"cited_work":{"arxiv_id":"2310.16828","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.16828","snapshot_observed_at":"2026-07-08T02:04:26.234596Z","title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","venue":"cs.LG","work_id":"360ec5fb-79fd-4490-bc73-3d161609c42d","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2310.16828","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:b3c6a329cb50067db0811bfd47842a9eae36eb99a7953fa2f14323b3d434fea5","observation_id":"b35811ad-c69e-4a7d-8415-fa3bcaf5986e","resolution":{"observed_at":"2026-05-14T17:27:36.461279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17080","last_updated":"2023-09-29T09:20:37Z","snapshot_observed_at":"2026-07-06T16:25:21.571679Z","submitted_at":"2023-09-29T09:20:37Z","title":"GAIA-1: A Generative World Model for Autonomous Driving","version":1},"cited_work":{"arxiv_id":"2309.17080","doi":null,"metadata_source":"pith","pith_arxiv_id":"2309.17080","snapshot_observed_at":"2026-07-10T11:47:02.949785Z","title":"GAIA-1: A Generative World Model for Autonomous Driving","venue":"cs.CV","work_id":"313484e6-a442-4522-8e19-d07e502844a8","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2309.17080","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:81e5e917268fcdda6297ec8d308b9838bf2dedc1e285e0746905e71da98876b3","observation_id":"c4ba759a-8199-4962-aa8e-2a77bd90c9d6","resolution":{"observed_at":"2026-05-12T07:15:10.779550Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2410.23506","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Huang, P., Liu, S., Liu, Z., Yan, Y ., Wang, S., Chen, Z., and Xiao, T","venue":null,"work_id":"1a9086d1-cebf-4c6a-b174-c63c34eb0ec9","year":null},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:8b61ab8d044b41c089083c1a7c52d086028cbff8b034681e4648ccd2ac69e7d5","observation_id":"05a596de-a1e0-4a57-8668-a476c90b7169","resolution":{"observed_at":"2026-05-11T00:33:50.897096Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.14795","last_updated":"2022-03-15T22:37:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-30T17:53:34Z","title":"Perceiver IO: A General Architecture for Structured Inputs & Outputs","version":3},"cited_work":{"arxiv_id":"2107.14795","doi":"10.48550/arxiv.2107.14795","metadata_source":"pith","pith_arxiv_id":"2107.14795","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Perceiver IO: A General Architecture for Structured Inputs & Outputs","venue":"cs.LG","work_id":"92bf7a73-2bef-4de4-8957-a9233d60b416","year":2021},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2107.14795","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:9f5830e1c03aa1409f5f58cd9103d54d6f2173216c97a82e4e752bdf960bd82b","observation_id":"a353802c-bc41-43c8-8ec1-ee2124de08e8","resolution":{"observed_at":"2026-05-15T19:47:14.368858Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.06950","last_updated":"2017-05-19T12:07:01Z","snapshot_observed_at":"2026-07-06T05:43:22.028213Z","submitted_at":"2017-05-19T12:07:01Z","title":"The Kinetics Human Action Video Dataset","version":1},"cited_work":{"arxiv_id":"1705.06950","doi":null,"metadata_source":"pith","pith_arxiv_id":"1705.06950","snapshot_observed_at":"2026-07-09T11:16:11.423695Z","title":"The Kinetics Human Action Video Dataset","venue":"cs.CV","work_id":"c8a3de61-cfd3-4aeb-bcf7-a0372c015748","year":2017},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1705.06950","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:c05a5ae6923e253476bc172f105342de73a06c5e41e8ae3afc2e56e0b6db46ed","observation_id":"512192e1-5c2f-421b-8717-7545025e78eb","resolution":{"observed_at":"2026-05-11T03:13:45.682066Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12945","last_updated":"2025-04-22T17:57:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-19T17:48:38Z","title":"DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset","version":2},"cited_work":{"arxiv_id":"2403.12945","doi":"10.48550/arxiv.2403.12945","metadata_source":"pith","pith_arxiv_id":"2403.12945","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset","venue":"cs.RO","work_id":"13253de2-3d89-415c-8c2f-3adb25d4c337","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2403.12945","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:82721373e2a5b4d4ef6afe5db20d27f97f9ec004f2a8c8d7a83bc0929754a133","observation_id":"d9daf3d7-f670-43fe-b71d-216708b2a2c3","resolution":{"observed_at":"2026-05-11T05:51:19.504346Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"cited_work":{"arxiv_id":"2406.09246","doi":"10.18653/v1/2022.naacl-main.68","metadata_source":"pith","pith_arxiv_id":"2406.09246","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","venue":"cs.RO","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2406.09246","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:dfd06b948ab696bf1caecf99dec161d04359336dd28d9b15dbf860015b7ad088","observation_id":"dee87fdd-3f34-4d1e-919a-a6fce3010e77","resolution":{"observed_at":"2026-05-11T00:33:50.935341Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A path towards autonomous machine intelligence version 0.9.2, 2022-06-27.Open Review, 62(1):1–62","venue":null,"work_id":"492cf134-9607-4764-a4bb-8a524ba68dd0","year":2022},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:b420434c77f9e7877c45fea38ba44e43b267d121c4224eb71b9fb6b901d8bf01","observation_id":"95317c8e-06fc-4c1f-bef4-3d1bc77394f1","resolution":{"observed_at":"2026-05-11T00:33:51.213849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:69ab5463469eb3baab6a8a3fae689b017d95c1e96bcb303ac8b0db9f575cad99","observation_id":"1a49219d-35ae-4aec-a482-1252cd056abb","resolution":{"observed_at":"2026-05-11T00:33:51.102812Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"cited_work":{"arxiv_id":"2403.00476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.00476","snapshot_observed_at":"2026-07-04T10:29:44.911015Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","venue":"cs.CV","work_id":"88f4d72e-ee93-420a-a61e-64b02b8cc454","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2403.00476","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:349fc1275fad5e351922260fc8cdd3500225799f02e02bedb012021edd357bfe","observation_id":"8359af7d-9ab3-4be8-9d3c-1fd8d95608af","resolution":{"observed_at":"2026-05-17T02:46:17.144047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":"1711.05101","doi":"10.1137/1.9781611972825.47","metadata_source":"pith","pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Decoupled Weight Decay Regularization","venue":"cs.LG","work_id":"07ef7360-d385-4033-83f7-8384a6325204","year":2017},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:facc8808324250b3ae02ee0ef1759a3d1b880941dc3bb500477ee3aa5e02474d","observation_id":"45cc6c29-ea7d-4df9-993c-173a43c8c513","resolution":{"observed_at":"2026-05-11T00:33:51.112985Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.05085","last_updated":"2020-09-10T18:38:10Z","snapshot_observed_at":"2026-08-05T13:44:37.867899Z","submitted_at":"2020-09-10T18:38:10Z","title":"Keypoints into the Future: Self-Supervised Correspondence in Model-Based Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2009.05085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2009.05085","snapshot_observed_at":"2026-06-29T22:13:59.837793Z","title":"Manuelli, Y","venue":null,"work_id":"221a4942-26fc-4c97-99c9-742c74cc1a8f","year":2009},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2009.05085","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:7db429af9098ca0e76f50a99967fb4d34f24465d8cc4c5a5ae0a6a381226641d","observation_id":"06953f0c-1aeb-41ad-a154-2c2e310e8590","resolution":{"observed_at":"2026-05-11T00:33:51.120897Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12213","last_updated":"2024-05-26T19:55:26Z","snapshot_observed_at":"2026-07-06T18:16:51.116432Z","submitted_at":"2024-05-20T17:57:01Z","title":"Octo: An Open-Source Generalist Robot Policy","version":2},"cited_work":{"arxiv_id":"2405.12213","doi":"10.48550/arxiv.2405.12213","metadata_source":"pith","pith_arxiv_id":"2405.12213","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Octo: An Open-Source Generalist Robot Policy","venue":"cs.RO","work_id":"f9ca0722-8855-48c3-a27a-0eefb7e19253","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2405.12213","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:b6ac28ba293f9a5e7e4dc93852e545d4fb775453881bf2358cd79304544ebb6b","observation_id":"9d623b42-e6a5-42be-9ef5-29b84742618d","resolution":{"observed_at":"2026-05-11T00:33:51.126566Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":"2304.07193","doi":"10.48550/arxiv.2304.07193","metadata_source":"pith","pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DINOv2: Learning Robust Visual Features without Supervision","venue":"cs.CV","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:f9e0769509d1c93ddd47204f50e8ac7e27cfec9e06a854a22a4319eb7a15f4d9","observation_id":"57365b87-d7e6-4990-aca8-c82a929eeb0a","resolution":{"observed_at":"2026-05-11T00:33:51.136489Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:624d96815ecf76757a338f08b6dc15beb135c778caefc0278f0aff39613678d8","observation_id":"06ea8c53-e396-447e-a2e7-7f7852559e51","resolution":{"observed_at":"2026-05-11T00:33:51.149057Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05453","last_updated":"2025-01-09T18:59:58Z","snapshot_observed_at":"2026-07-06T20:18:53.977126Z","submitted_at":"2025-01-09T18:59:58Z","title":"An Empirical Study of Autoregressive Pre-training from Videos","version":1},"cited_work":{"arxiv_id":"2501.05453","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.05453","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2501.05453 , year=","venue":null,"work_id":"b62ba86b-2d51-47fd-bcfe-f8919a8a3cac","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2501.05453","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:9a0669ebf6f20508c7ffb37249699be39769eda256ae838576553005cce55734","observation_id":"93043d74-8061-46d9-9ec8-2118051a49ef","resolution":{"observed_at":"2026-05-11T00:33:51.155917Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20523","last_updated":"2025-03-26T13:11:35Z","snapshot_observed_at":"2026-07-06T20:59:00.415760Z","submitted_at":"2025-03-26T13:11:35Z","title":"GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving","version":1},"cited_work":{"arxiv_id":"2503.20523","doi":"10.48550/arxiv.2511.09057","metadata_source":"pith","pith_arxiv_id":"2503.20523","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving","venue":"cs.CV","work_id":"1339e674-d09b-48b4-8e6f-efe55dcab22e","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2503.20523","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:35f6c5865e1bbb7d8cc8cd045525feaec2dfd57a35c7ce17d9e156e8bdb67b49","observation_id":"d57d4997-31f2-4682-a732-3cee5148dacf","resolution":{"observed_at":"2026-05-15T13:48:22.447793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.23266","last_updated":"2025-08-25T17:57:34Z","snapshot_observed_at":"2026-07-06T19:42:25.742756Z","submitted_at":"2024-10-30T17:50:23Z","title":"TOMATO: Assessing Visual Temporal Reasoning Capabilities in Multimodal Foundation Models","version":2},"cited_work":{"arxiv_id":"2410.23266","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.23266","snapshot_observed_at":"2026-07-03T05:57:41.270398Z","title":"Tomato: Assessing visual temporal reasoning capabilities in multimodal foundation models","venue":null,"work_id":"7a50d850-6338-479d-ba02-65efc1b03f01","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2410.23266","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:e5cdd292866cc5779c3268ea8127b3bde6dec325e9ae0250a6933a14df4d18af","observation_id":"dd1c37b8-704b-4224-888d-0992d7313db4","resolution":{"observed_at":"2026-05-11T00:33:51.178602Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.14819","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:39:45.952845Z","title":"Learning from reward-free offline data: A case for planning with latent dynamics models","venue":null,"work_id":"f830428f-0092-4ff4-9c21-50f5a214b7d8","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:02f1093d9683b1a36ea3e7fbe336c97866997183c6eafae23a4f4c0182c1a0e0","observation_id":"f501c319-0753-418f-8e25-5100c409fa1e","resolution":{"observed_at":"2026-05-11T00:33:51.190282Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.09533","last_updated":"2024-06-25T17:57:38Z","snapshot_observed_at":"2026-07-06T18:45:35.689341Z","submitted_at":"2024-06-25T17:57:38Z","title":"Video Occupancy Models","version":1},"cited_work":{"arxiv_id":"2407.09533","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.09533","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"E., and Levine, S","venue":null,"work_id":"9f738baa-831c-42b5-b186-6d93934a45ac","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2407.09533","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:b2035a1d65f897bbda2eb62b8ab1dc1288b7ae58f5c30d7c1dc6ae7abf5b7c2e","observation_id":"ed8e984d-368c-4623-95b6-256643006d18","resolution":{"observed_at":"2026-05-11T00:33:51.198027Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14786","last_updated":"2025-02-20T18:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-20T18:08:29Z","title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","version":1},"cited_work":{"arxiv_id":"2502.14786","doi":"10.48550/arxiv.2502.14786","metadata_source":"pith","pith_arxiv_id":"2502.14786","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","venue":"cs.CV","work_id":"50eec732-2d41-432f-9dcf-ac7fff235ea5","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2502.14786","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:9c75810a1b225d07d61e4b95757f63bf2c2ba99828e56922ce0afd609dd42f59","observation_id":"eec793d8-d0c9-4b3f-b870-7fcde67106e0","resolution":{"observed_at":"2026-05-11T00:33:51.202552Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-10T23:49:08.777694+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T23:49:08.777694+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17927","last_updated":"2024-05-28T07:48:15Z","snapshot_observed_at":"2026-08-05T05:26:20.143730Z","submitted_at":"2024-05-28T07:48:15Z","title":"The Evolution of Multimodal Model Architectures","version":1},"cited_work":{"arxiv_id":"2405.17927","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.17927","snapshot_observed_at":"2026-06-30T23:45:08.065399Z","title":"Junyang Wang, Yuhang Wang, Guohai Xu, Jing Zhang, Yukai Gu, Haitao Jia, Jiaqi Wang, Haiyang Xu, Ming Yan, Ji Zhang, et al","venue":null,"work_id":"a8b21d3b-d291-4986-8d2a-70ddf9fc7c2f","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2405.17927","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:e160f8be04cf22714f2b7ca139b74897852825ffe20489ddb295bba0e858f305","observation_id":"03b7c11b-16a0-4b97-8a9c-c4fa9e3062d9","resolution":{"observed_at":"2026-05-11T00:33:50.954145Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":"2409.12191","doi":"10.48550/arxiv.2409.12191","metadata_source":"pith","pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","venue":"cs.CV","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:89458b96fe22d9476383756abb61cf649b0689a715b0a3e642cc0bc6bc4cf84f","observation_id":"efc3332f-1f50-491c-85bb-3c960ecc534a","resolution":{"observed_at":"2026-05-11T00:33:50.959202Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13139","last_updated":"2023-12-21T05:34:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-20T16:00:43Z","title":"Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation","version":2},"cited_work":{"arxiv_id":"2312.13139","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.13139","snapshot_observed_at":"2026-07-04T21:00:09.556391Z","title":"Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation","venue":"cs.RO","work_id":"e92c2c13-4330-45fe-8231-34a6002626bd","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2312.13139","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:36fb475526ecadd738d0b5c916c67acf9ac64819826dca8cd3612594f4ff3008","observation_id":"0ea17cc3-a4f7-4b31-8377-a22e27bd17ec","resolution":{"observed_at":"2026-05-13T16:32:05.974462Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07888","last_updated":"2025-01-24T05:16:36Z","snapshot_observed_at":"2026-07-06T20:20:47.146343Z","submitted_at":"2025-01-14T06:54:39Z","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","version":3},"cited_work":{"arxiv_id":"2501.07888","doi":"10.48550/arxiv.2501.07888","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.07888","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tarsier2: Advancing large vision-language models from detailed video description to comprehensive video understanding","venue":"ArXiv.org","work_id":"77605c65-9b44-4521-ad8c-79d9c78cf33a","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2501.07888","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:b6882a7ca50e20eaf850424ba732afc0cb8e0070f7a17aeb99b57cdb8a7794d8","observation_id":"506152f2-e699-4975-852f-f94d86af54a0","resolution":{"observed_at":"2026-05-11T00:33:50.994322Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":"2306.02858","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-07-04T16:29:57.761121Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","venue":"cs.CL","work_id":"555cf04a-49a7-44b8-9019-a83ce85ace95","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:ce96c80525d177d379d31f1ac6a539053bef10d29ede79acb1dc3648a17cfcf2","observation_id":"5f53b7b9-dd40-42ca-9226-7639e797b152","resolution":{"observed_at":"2026-05-13T15:02:01.218694Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12772","last_updated":"2025-05-05T04:48:45Z","snapshot_observed_at":"2026-07-06T18:47:56.109836Z","submitted_at":"2024-07-17T17:51:53Z","title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","version":2},"cited_work":{"arxiv_id":"2407.12772","doi":"10.48550/arxiv.2407.12772","metadata_source":"pith","pith_arxiv_id":"2407.12772","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","venue":"cs.CL","work_id":"257da118-790b-4686-87d7-92321d7e1ae0","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2407.12772","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:e743e94f9f5e84510193d65402bd602ce6e7b621b02b66ee270013ef8971b586","observation_id":"15bd45bc-a1b8-4c77-8dd8-83626e44a128","resolution":{"observed_at":"2026-05-17T05:19:22.513527Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.22020","last_updated":"2025-03-27T22:23:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-27T22:23:04Z","title":"CoT-VLA: Visual Chain-of-Thought Reasoning for Vision-Language-Action Models","version":1},"cited_work":{"arxiv_id":"2503.22020","doi":"10.48550/arxiv.2503.22020","metadata_source":"pith","pith_arxiv_id":"2503.22020","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoT-VLA: Visual Chain-of-Thought Reasoning for Vision-Language-Action Models","venue":"cs.CV","work_id":"14a3cf4d-f3aa-46dd-a613-bb5253154921","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2503.22020","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:faba95e8a364e7aea734ef2c39d4bad3ed9e71675d2d3147cccdf58399daadbf","observation_id":"f937889c-ac7f-4d16-b967-bec5f6d8c96a","resolution":{"observed_at":"2026-05-16T05:21:45.399848Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15659","last_updated":"2025-05-21T15:33:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T15:33:27Z","title":"FLARE: Robot Learning with Implicit World Modeling","version":1},"cited_work":{"arxiv_id":"2505.15659","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15659","snapshot_observed_at":"2026-07-10T12:47:05.520966Z","title":"FLARE: Robot Learning with Implicit World Modeling","venue":"cs.RO","work_id":"0734908a-7122-4eeb-ba5d-944092bb8897","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2505.15659","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:4cb12c923a73cbffd46d707c963c075031c4da435e57ae72a3e670ec1d16cfc1","observation_id":"5ad1e2d9-6c37-401d-bb2c-b32321d5c289","resolution":{"observed_at":"2026-05-17T15:59:09.111136Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04983","last_updated":"2025-02-01T02:40:49Z","snapshot_observed_at":"2026-07-06T19:46:54.707852Z","submitted_at":"2024-11-07T18:54:37Z","title":"DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning","version":2},"cited_work":{"arxiv_id":"2411.04983","doi":"10.48550/arxiv.2411.04983","metadata_source":"pith","pith_arxiv_id":"2411.04983","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning","venue":"cs.RO","work_id":"4a946586-a786-46da-9388-197c5410bf39","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2411.04983","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:0dc1e21c62427a82a0e42f60fd1bde7e9d2a4d7e4c9e7c8ccfad16d8cdab6e04","observation_id":"339aa09a-d664-4b98-98dd-6f5afc436f7f","resolution":{"observed_at":"2026-05-17T16:06:10.055251Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-11T22:49:30.554473+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T22:49:30.554473+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02792","last_updated":"2025-05-23T00:47:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-03T17:38:59Z","title":"Unified World Models: Coupling Video and Action Diffusion for Pretraining on Large Robotic Datasets","version":3},"cited_work":{"arxiv_id":"2504.02792","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.02792","snapshot_observed_at":"2026-07-10T18:47:31.709153Z","title":"Unified World Models: Coupling Video and Action Diffusion for Pretraining on Large Robotic Datasets","venue":"cs.RO","work_id":"c181dcf1-e774-4216-a30e-e55c3f3a766c","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"cited_paper":"/paper/2504.02792","citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:aa53832de460522b1b4d266d14c8d773074db2644067eab083a8862883eac395","observation_id":"93e6b606-7cf2-4cff-936d-0abc15437426","resolution":{"observed_at":"2026-05-13T16:25:00.606134Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"abbreviated","venue":null,"work_id":"2e487ef0-7a23-4c9a-b52d-0d8b7a9ab1cd","year":2024},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:8ff25124526bb578659f9291a1528a8230315aa9baa89c44745bce0c066c9397","observation_id":"a1188a44-03fc-4259-831f-f1f75f816862","resolution":{"observed_at":"2026-05-11T00:33:51.234814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"(2020), using the standard16 × 16 patch size","venue":null,"work_id":"35a761b6-2db7-4d15-8c10-d9a549f8b258","year":2020},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:8f4d6ecd036360d5f1260028e35b240bb29b89aab710cb88a5276e73e7d47240","observation_id":"9652bef8-da87-4409-b4ff-cbec3f69077c","resolution":{"observed_at":"2026-05-11T00:33:51.250197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"For the pick-and-place tasks we present two sub-goal images to the model in addition to the final goal","venue":null,"work_id":"8573fdab-4d42-447a-af5b-d1e34af524d2","year":2000},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:6af9b8307ef065ffd715efe26856a64e8c8437408e94b537d5d4a7b4cba71aa1","observation_id":"1df189b3-4176-4803-8df5-004026c7a7ba","resolution":{"observed_at":"2026-05-11T00:33:51.255541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The MLLM ingests the output embeddings of the vision encoder, which are projected to the hidden dimension of the LLM backbone using aprojector module","venue":null,"work_id":"b3b96e01-943c-4f2c-b459-79aab071b837","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:78fcc749b7c540983fb871a4b0ae41d568f29470c70ba277fd54508dc28582da","observation_id":"3cca2a2f-f3c5-47cb-bfcf-cb6bbd4d06a8","resolution":{"observed_at":"2026-05-11T00:33:51.261444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We describe the training details in the following sections","venue":null,"work_id":"e17334cf-6e41-4b4b-bb9c-c3f0d29c3966","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:e03387e122fbaef7b2e5caf70f9db369bd3d1328e76f18096c5d66125c686f0b","observation_id":"763b4da0-03cc-4965-9bad-c241fb7b9ac3","resolution":{"observed_at":"2026-05-11T00:33:51.217765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"To assess the ability of V-JEPA 2 to capture spatiotemporal details for VidQA, we compare to leading off-shelf image encoders","venue":null,"work_id":"e9a66a4b-fa8d-428c-8b44-d30d5547fee8","year":2023},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:63a6eb0ce97feb1b43902d89678d05ec2a5a363c5d86f5e63bca42fa0377776b","observation_id":"98a591e2-65dc-4d12-aa70-daabf6b9780d","resolution":{"observed_at":"2026-05-11T00:33:51.221451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unlike Cho et al","venue":null,"work_id":"71e1dd4f-f5e6-41b6-a387-4574175cd379","year":2025},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:c2502161774d7d33c42ea4d78c03608ffd82c82b46d752b3c6da7a527c992a24","observation_id":"689f5c30-4102-4141-a14d-e6b9b166681a","resolution":{"observed_at":"2026-05-11T00:33:51.224273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We scale up the data size to 88.5 million samples","venue":null,"work_id":"e75c3c6e-65ec-4b5e-8cda-e676f396621d","year":2048},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:f90579b0ff66a29914817762809a38b5e396bcac734e2ce17cc9aa4dce61a958","observation_id":"8c6206a5-0347-4ae4-a5f0-65cce4f07999","resolution":{"observed_at":"2026-05-11T00:33:51.227970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1c92a40a-ecb9-47f2-ba33-23db8d1eb54c","year":1920},"citing_paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-11T00:33:50.471804Z"},"links":{"citing_paper":"/paper/2506.09985"},"observation_digest":"sha256:4bad41871c1c27fec25ccb2ecbbf174c82cf61258f7ae339eea87e604773618a","observation_id":"5766f544-a493-431f-bb50-22e375a8bae9","resolution":{"observed_at":"2026-05-11T00:33:51.230791Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.09985","last_updated":"2025-06-11T17:57:09Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-11T17:57:09Z","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning"},"reference_resolution":{"displayed":69,"state_counts":{"malformed_identifier":0,"metadata_mismatch":11,"parse_uncertain":0,"unresolved":1,"verified_exact":48,"verified_fuzzy":9},"total_outbound_references":69},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 69 of 69 outbound references and 100 inbound Pith citation observations for arXiv:2506.09985."}