{"as_of":"2026-08-15T14:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3ffc25f456a9ceebc07555202b0e65e7262049216e4bf86f517441d0b98fabac","coverage":[{"denominator":16,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T15:33:35.034268Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:37:35.343073Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T15:49:58.016388Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10925","snapshot_observed_at":"2026-08-07T15:37:35.343073Z","title":"Video representation learning with joint-embedding predictive architectures","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14543","last_updated":"2025-05-20T15:58:54Z","snapshot_observed_at":"2026-08-13T16:30:28.776060Z","submitted_at":"2025-05-20T15:58:54Z","title":"Time to Embed: Unlocking Foundation Models for Time Series with Channel Descriptions","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T15:37:35.343073Z"},"links":{"cited_paper":"/paper/2412.10925","citing_paper":"/paper/2505.14543"},"observation_digest":"sha256:7c409cd7c69a5bf3132c0d31830524900d56f46fde07931c505c5fe6eabaad06","observation_id":"99be13f5-de93-4de8-aef3-531f3a746e39","resolution":{"observed_at":"2026-08-07T15:37:35.343073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10925","snapshot_observed_at":"2026-08-06T21:09:16.043159Z","title":"Video representation learning with joint-embedding predictive architectures,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00917","last_updated":"2025-09-03T01:44:58Z","snapshot_observed_at":"2026-08-15T04:49:35.866914Z","submitted_at":"2025-07-01T16:23:00Z","title":"A Survey: Learning Embodied Intelligence from Physical Simulators and World Models","version":3},"reference_index":266,"source":"pdf_text","source_observed_at":"2026-08-06T21:09:16.043159Z"},"links":{"cited_paper":"/paper/2412.10925","citing_paper":"/paper/2507.00917"},"observation_digest":"sha256:e0cbedbdb4c56c4f3e191b03478b93417852b87dbc15be7ddcaeb868502d86dd","observation_id":"4119aa35-b59e-48b4-999e-0424ccc17b56","resolution":{"observed_at":"2026-08-06T21:09:16.043159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"cited_work":{"arxiv_id":"2412.10925","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.10925","snapshot_observed_at":"2026-07-04T15:49:58.016388Z","title":"arXiv preprint arXiv:2412.10925 , year=","venue":null,"work_id":"a69650fc-9f6a-412e-a5d9-811d43a79a9d","year":2024},"citing_paper":{"arxiv_id":"2605.05586","last_updated":"2026-05-07T02:11:26Z","snapshot_observed_at":"2026-08-11T03:55:06.783509Z","submitted_at":"2026-05-07T02:11:26Z","title":"AeroJEPA: Learning Semantic Latent Representations for Scalable 3D Aerodynamic Field Modeling","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-09T15:53:39.074115Z"},"links":{"cited_paper":"/paper/2412.10925","citing_paper":"/paper/2605.05586"},"observation_digest":"sha256:e4d6a139347f3fc1cc9741fd56357afdddd745e0b099043419c5f66055c2831c","observation_id":"8f9a0b6f-ffd7-4153-9a00-be14658ebe90","resolution":{"observed_at":"2026-05-11T16:36:09.367752Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"cited_work":{"arxiv_id":"2412.10925","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.10925","snapshot_observed_at":"2026-07-04T15:49:58.016388Z","title":"arXiv preprint arXiv:2412.10925 , year=","venue":null,"work_id":"a69650fc-9f6a-412e-a5d9-811d43a79a9d","year":2024},"citing_paper":{"arxiv_id":"2606.26410","last_updated":"2026-06-24T22:06:35Z","snapshot_observed_at":"2026-08-07T03:54:49.094588Z","submitted_at":"2026-06-24T22:06:35Z","title":"Neural Voxel Dynamics: Learning Implicit 3D Physics via Volumetric Feature Advection","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-26T01:17:06.587807Z"},"links":{"cited_paper":"/paper/2412.10925","citing_paper":"/paper/2606.26410"},"observation_digest":"sha256:1d21464496310abbbe19e26c4d656eb9c957dbb5ad0e589d915110ae6b1bb522","observation_id":"46b3717d-678d-4112-bd29-144f9f242670","resolution":{"observed_at":"2026-07-04T15:49:58.017770Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2412.10925/citation-record","integrity":"/paper/2412.10925/integrity","json":"/paper/2412.10925/citation-record.json","paper":"/paper/2412.10925"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-08-14T18:51:16.666127Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-11T15:33:34.746792Z","title":"Adam: A method for stochastic optimization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.746792Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:ba1dee61e555bd4486f0df7d6576cc908b6371b47a6e4db8c65192186be157dd","observation_id":"43a3a51e-1fea-4970-8d14-b650079c9fd7","resolution":{"observed_at":"2026-08-11T15:33:34.746792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.03248","last_updated":"2019-06-07T17:22:54Z","snapshot_observed_at":"2026-08-14T16:19:04.978942Z","submitted_at":"2019-06-07T17:22:54Z","title":"Evolving Losses for Unlabeled Video Representation Learning","version":1},"cited_work":{"arxiv_id":"1906.03248","doi":null,"metadata_source":"pith","pith_arxiv_id":"1906.03248","snapshot_observed_at":"2026-08-11T15:33:35.102241Z","title":"Evolving Losses for Unlabeled Video Representation Learning","venue":"cs.CV","work_id":"f0481936-7808-4304-8b45-5988d904c8b9","year":2019},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.938748Z"},"links":{"cited_paper":"/paper/1906.03248","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:33fd45112abefde918b0414e0936c9efa82431f141929fdd10d27838c4bc6b91","observation_id":"e6bc7076-ee9e-4933-a28b-511a6fced05b","resolution":{"observed_at":"2026-08-11T15:33:35.253458Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.00633","last_updated":"2024-05-02T03:58:14Z","snapshot_observed_at":"2026-08-15T13:41:37.101736Z","submitted_at":"2023-03-01T16:36:25Z","title":"An Information-Theoretic Perspective on Variance-Invariance-Covariance Regularization","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.00633","snapshot_observed_at":"2026-08-11T15:33:34.964758Z","title":"An information- theoretic perspective on variance-invariance-covariance regularization.arXiv preprint arXiv:2303.00633 ,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.964758Z"},"links":{"cited_paper":"/paper/2303.00633","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:0dd51629036199d028964a504a815d266f9bf53f04a7919dbf67c4f09e5b857f","observation_id":"c11d5329-9947-4222-be01-4c3990ebc822","resolution":{"observed_at":"2026-08-11T15:33:34.964758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01442","last_updated":"2020-03-08T00:09:07Z","snapshot_observed_at":"2026-07-06T08:26:38.349660Z","submitted_at":"2019-10-03T13:16:36Z","title":"CLEVRER: CoLlision Events for Video REpresentation and Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.01442","snapshot_observed_at":"2026-08-11T15:33:34.998477Z","title":"Clevrer: Collision events for video representation and reasoning.arXiv preprint arXiv:1910.01442 ,","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.998477Z"},"links":{"cited_paper":"/paper/1910.01442","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:2df143c160f06fba390c09dc80e5b0fe06a6ddd7004bdb9bad41c5b81198c768","observation_id":"5c63c0fe-e37a-4732-a3e2-af880b1cc40e","resolution":{"observed_at":"2026-08-11T15:33:34.998477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13292","last_updated":"2024-02-22T21:07:10Z","snapshot_observed_at":"2026-08-13T11:10:52.301161Z","submitted_at":"2023-06-23T05:01:02Z","title":"Variance-Covariance Regularization Improves Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13292","snapshot_observed_at":"2026-08-11T15:33:35.020514Z","title":"Variance-covariance regularization improves representation learning.arXiv preprint arXiv:2306.13292 ,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:35.020514Z"},"links":{"cited_paper":"/paper/2306.13292","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:661c5025540022290845bd90e70b806dbff90cc13b657574fc04fddaa93fa290","observation_id":"511f9f14-6c56-43fe-9fb1-0790c3881300","resolution":{"observed_at":"2026-08-11T15:33:35.020514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:35.573856Z","title":null,"venue":null,"work_id":"0b3e87b9-8662-44d6-835b-ffae3a9fc796","year":2019},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:35.026041Z"},"links":{"citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:cfe09acdc66e8fa07ee0952c0afb4d87f68627ab08ff8b340babbd76437d62b9","observation_id":"e6c97888-4f75-4dde-8177-38c0eb58ea32","resolution":{"observed_at":"2026-08-11T15:33:35.577831Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:35.560652Z","title":"Each video contains 300 frames at 24 frames per second at 320x240 resolution","venue":null,"work_id":"25844a8d-0b13-48c5-87d0-3ce16dcf716e","year":2019},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:35.030323Z"},"links":{"citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:0ba7eb89cddd89c96581add6ae897a523c5d2a65847a835d97a84f5f217168e0","observation_id":"3ae96479-0343-4154-8efe-6ad4cac5a881","resolution":{"observed_at":"2026-08-11T15:33:35.564661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:35.382316Z","title":"The model on the left has PSNR of 22.8 and the one on the right has PSNR of 21.2","venue":null,"work_id":"15f13ba9-19dd-4b60-8e95-5aab6a589625","year":2019},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:35.034268Z"},"links":{"citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:1109ba835d3161031838f8b4bd0facc583951c3cee9653dec58a70a11f5e699a","observation_id":"9c4a4b5b-e4ea-438f-b450-8482bcfe85ec","resolution":{"observed_at":"2026-08-11T15:33:35.533152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:35.895429Z","title":"neurips.cc/paper_files/paper/1993/file/288cc0ff022877bd3df94bc9360b9c5d-Paper.pdf","venue":null,"work_id":"9362b672-dd20-4188-b11b-60ad2cb1149c","year":1993},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":1993,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.633727Z"},"links":{"citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:50e533ec312a010e10003258d1612987bf8e13d2715ce20d39413c995ecebf92","observation_id":"2b1bb1f0-92fb-4a67-9fbd-d791367c1a04","resolution":{"observed_at":"2026-08-11T15:33:35.901120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:35.586255Z","title":"A path towards autonomous machine intelligence version 0.9","venue":null,"work_id":"01552ab6-aac1-484a-9f51-6f4ce027464a","year":2022},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.822235Z"},"links":{"citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:ae13a0ab699908889146d6061b67610fc5c3a715a2a5df123dc42c0c23267b9a","observation_id":"cd786fd0-65b6-47bf-8cb6-dbda02d415cb","resolution":{"observed_at":"2026-08-11T15:33:35.590337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.12808","last_updated":"2022-12-24T19:44:41Z","snapshot_observed_at":"2026-08-13T16:24:25.322639Z","submitted_at":"2022-12-24T19:44:41Z","title":"A Comprehensive Review on Autonomous Navigation","version":1},"cited_work":{"arxiv_id":"2212.12808","doi":null,"metadata_source":"pith","pith_arxiv_id":"2212.12808","snapshot_observed_at":"2026-08-11T15:33:35.306220Z","title":"A Comprehensive Review on Autonomous Navigation","venue":"cs.RO","work_id":"a0e6b4b7-e469-4b55-8fe9-913858b67392","year":2022},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.872349Z"},"links":{"cited_paper":"/paper/2212.12808","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:e5ab53a16e1212971d105744c2c9df15d9abc75de9e32c9782149c02878060c4","observation_id":"4512778e-7f4d-4259-a0ba-9404a5e88499","resolution":{"observed_at":"2026-08-11T15:33:35.311554Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.09348","last_updated":"2022-04-23T16:44:20Z","snapshot_observed_at":"2026-08-15T03:32:28.162275Z","submitted_at":"2021-10-18T14:22:19Z","title":"Understanding Dimensional Collapse in Contrastive Self-supervised Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.09348","snapshot_observed_at":"2026-08-11T15:33:34.723038Z","title":"Understanding dimensional collapse in contrastive self-supervised learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.723038Z"},"links":{"cited_paper":"/paper/2110.09348","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:8161c7b1aea3b6cc876da980d0126992ecd9ee2dc9dd45d8a83cf215d62a10db","observation_id":"59569077-ed74-432d-bad6-9d9b22838424","resolution":{"observed_at":"2026-08-11T15:33:34.723038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.04994","last_updated":"2017-11-30T23:11:58Z","snapshot_observed_at":"2026-08-15T06:25:24.915298Z","submitted_at":"2017-11-14T08:32:43Z","title":"Prediction Under Uncertainty with Error-Encoding Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.04994","snapshot_observed_at":"2026-08-11T15:33:34.718900Z","title":"Prediction under uncertainty with error-encoding networks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.718900Z"},"links":{"cited_paper":"/paper/1711.04994","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:58a5fe5ff3a46dc062524f1848bf3a803f6d81495eaff9bbc005ad7492fd509a","observation_id":"2aa3a84e-aebd-407a-b62d-46d5d3986f52","resolution":{"observed_at":"2026-08-11T15:33:34.718900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:33:35.673238Z","title":"Dimensionality reduction by learning an invariant mapping","venue":null,"work_id":"159820f5-59ef-45d2-8ef0-ecf5345c1464","year":2006},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.714554Z"},"links":{"citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:76daea7eb537a06a1ae12d762b345d00c5e009d9b9ca8c7eddee57abfb5b7de0","observation_id":"cb4c7050-19b8-4581-8495-d89385d36385","resolution":{"observed_at":"2026-08-11T15:33:35.800813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.05440","last_updated":"2016-02-26T22:10:30Z","snapshot_observed_at":"2026-08-14T22:22:34.655459Z","submitted_at":"2015-11-17T15:36:32Z","title":"Deep multi-scale video prediction beyond mean square error","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1511.05440","snapshot_observed_at":"2026-08-11T15:33:34.828546Z","title":"Deep multi-scale video prediction beyond mean square error","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.828546Z"},"links":{"cited_paper":"/paper/1511.05440","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:5799f8050608d38a03a420f9df37d16231b333ef50f7e184394f3abb7aa3ab5d","observation_id":"1249f408-1edf-44fa-a101-ed43c70cdf8f","resolution":{"observed_at":"2026-08-11T15:33:34.828546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.04744","last_updated":"2020-04-05T03:39:21Z","snapshot_observed_at":"2026-08-13T13:07:55.618967Z","submitted_at":"2019-10-10T17:52:19Z","title":"CATER: A diagnostic dataset for Compositional Actions and TEmporal Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.04744","snapshot_observed_at":"2026-08-11T15:33:34.688128Z","title":"Cater: A diagnostic dataset for compositional actions and temporal reasoning","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-11T15:33:34.688128Z"},"links":{"cited_paper":"/paper/1910.04744","citing_paper":"/paper/2412.10925"},"observation_digest":"sha256:b083cc6e7c8ac61376d6b216666fd66046c298fce74d5f600e12e385187896ff","observation_id":"13911945-1afd-4b63-80e6-a897e2f06dfb","resolution":{"observed_at":"2026-08-11T15:33:34.688128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.10925","last_updated":"2024-12-14T18:33:29Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-11T16:50:42.881971Z","submitted_at":"2024-12-14T18:33:29Z","title":"Video Representation Learning with Joint-Embedding Predictive Architectures"},"reference_resolution":{"displayed":16,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":2,"verified_fuzzy":5},"total_outbound_references":16},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 16 of 16 outbound references and 4 inbound Pith citation observations for arXiv:2412.10925."}