{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CE3YOYL24LTV265ET7MPOONM6P","short_pith_number":"pith:CE3YOYL2","schema_version":"1.0","canonical_sha256":"113787617ae2e75d7ba49fd8f739acf3e25eb0c7ff2fee08c1960a749b2fa2e1","source":{"kind":"arxiv","id":"2412.15212","version":2},"attestation_state":"computed","paper":{"title":"Scaling 4D Representations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Andrew Zisserman, Aravindh Mahendran, Carl Doersch, Chris Duvarney, Chuhan Zhang, Daniel Zoran, Dilara Gokay, Dima Damen, Drew A. Hudson, Eric Aboussouan, Etienne Pot, Goker Erdogan, Guillaume Le Moing, Ignacio Rocco, Jacob Walker, Jennifer Sun, Jo\\~ao Carreira, Joseph Heyward, Kelsey Allen, Klaus Greff, Luisa Polan\\'ia, Luke Friedman, Mehdi S. M. Sajjadi, Michael King, Pauline Luc, Pedro V\\'elez, Rishabh Kabra, Ross Goroshin, Sjoerd van Steenkiste, Skanda Koppula, Thomas Albert Keck, Thomas Kipf, Viorica P\\u{a}tr\\u{a}ucean, Yana Hasson, Yi Yang","submitted_at":"2024-12-19T18:59:51Z","abstract_excerpt":"Scaling has not yet been convincingly demonstrated for pure self-supervised learning from video. However, prior work has focused evaluations on semantic-related tasks $\\unicode{x2013}$ action classification, ImageNet classification, etc. In this paper we focus on evaluating self-supervised learning on non-semantic vision tasks that are more spatial (3D) and temporal (+1D = 4D), such as camera pose estimation, point and object tracking, and depth estimation. We show that by learning from very large video datasets, masked auto-encoding (MAE) with transformer video models actually scales, consist"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15212","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-19T18:59:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"78484474e93f87a5916ea726c5d833f7e7189b4b4afc10d8e2879b11f5200c19","abstract_canon_sha256":"75b97c8c790139a0250e23a68db752ed1d96c2c9b09de89d15b5aff24d27f7d6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:02.993471Z","signature_b64":"8ZtB9HIXY7EU/bQnjYHsOUveKfMMGKJ5hksN+l32GMwfs6RaCHGvgaiylLdNriX7k/qyEx9nB41b7Z1z7sMIDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"113787617ae2e75d7ba49fd8f739acf3e25eb0c7ff2fee08c1960a749b2fa2e1","last_reissued_at":"2026-07-05T11:34:02.992910Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:02.992910Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling 4D Representations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Andrew Zisserman, Aravindh Mahendran, Carl Doersch, Chris Duvarney, Chuhan Zhang, Daniel Zoran, Dilara Gokay, Dima Damen, Drew A. Hudson, Eric Aboussouan, Etienne Pot, Goker Erdogan, Guillaume Le Moing, Ignacio Rocco, Jacob Walker, Jennifer Sun, Jo\\~ao Carreira, Joseph Heyward, Kelsey Allen, Klaus Greff, Luisa Polan\\'ia, Luke Friedman, Mehdi S. M. Sajjadi, Michael King, Pauline Luc, Pedro V\\'elez, Rishabh Kabra, Ross Goroshin, Sjoerd van Steenkiste, Skanda Koppula, Thomas Albert Keck, Thomas Kipf, Viorica P\\u{a}tr\\u{a}ucean, Yana Hasson, Yi Yang","submitted_at":"2024-12-19T18:59:51Z","abstract_excerpt":"Scaling has not yet been convincingly demonstrated for pure self-supervised learning from video. However, prior work has focused evaluations on semantic-related tasks $\\unicode{x2013}$ action classification, ImageNet classification, etc. In this paper we focus on evaluating self-supervised learning on non-semantic vision tasks that are more spatial (3D) and temporal (+1D = 4D), such as camera pose estimation, point and object tracking, and depth estimation. We show that by learning from very large video datasets, masked auto-encoding (MAE) with transformer video models actually scales, consist"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15212","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15212/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15212","created_at":"2026-07-05T11:34:02.992969+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15212v2","created_at":"2026-07-05T11:34:02.992969+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15212","created_at":"2026-07-05T11:34:02.992969+00:00"},{"alias_kind":"pith_short_12","alias_value":"CE3YOYL24LTV","created_at":"2026-07-05T11:34:02.992969+00:00"},{"alias_kind":"pith_short_16","alias_value":"CE3YOYL24LTV265E","created_at":"2026-07-05T11:34:02.992969+00:00"},{"alias_kind":"pith_short_8","alias_value":"CE3YOYL2","created_at":"2026-07-05T11:34:02.992969+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06856","citing_title":"Gen4U: Unifying Video Generation and Understanding via Diffusion","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03837","citing_title":"Where Do We (Not) Need Temporal Context in Low-Resource Video Task Adaptation?","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19137","citing_title":"Towards Data-Efficient Video Pre-training with Frozen Image Foundation Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.13942","citing_title":"Frozen Forecasting: A Unified Evaluation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27448","citing_title":"LA-Pose: Latent Action Pretraining Meets Pose Estimation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09985","citing_title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P","json":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P.json","graph_json":"https://pith.science/api/pith-number/CE3YOYL24LTV265ET7MPOONM6P/graph.json","events_json":"https://pith.science/api/pith-number/CE3YOYL24LTV265ET7MPOONM6P/events.json","paper":"https://pith.science/paper/CE3YOYL2"},"agent_actions":{"view_html":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P","download_json":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P.json","view_paper":"https://pith.science/paper/CE3YOYL2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15212&json=true","fetch_graph":"https://pith.science/api/pith-number/CE3YOYL24LTV265ET7MPOONM6P/graph.json","fetch_events":"https://pith.science/api/pith-number/CE3YOYL24LTV265ET7MPOONM6P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P/action/storage_attestation","attest_author":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P/action/author_attestation","sign_citation":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P/action/citation_signature","submit_replication":"https://pith.science/pith/CE3YOYL24LTV265ET7MPOONM6P/action/replication_record"}},"created_at":"2026-07-05T11:34:02.992969+00:00","updated_at":"2026-07-05T11:34:02.992969+00:00"}