{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:4OT3VPEVHLWWUBJ3VGW44ZLJP5","short_pith_number":"pith:4OT3VPEV","schema_version":"1.0","canonical_sha256":"e3a7babc953aed6a053ba9adce65697f6794aa9bc6ca98f472122adb3c9ffa71","source":{"kind":"arxiv","id":"2203.06173","version":1},"attestation_state":"computed","paper":{"title":"Masked Visual Pre-training for Motor Control","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Ilija Radosavovic, Jitendra Malik, Tete Xiao, Trevor Darrell","submitted_at":"2022-03-11T18:58:10Z","abstract_excerpt":"This paper shows that self-supervised visual pre-training from real-world images is effective for learning motor control tasks from pixels. We first train the visual representations by masked modeling of natural images. We then freeze the visual encoder and train neural network controllers on top with reinforcement learning. We do not perform any task-specific fine-tuning of the encoder; the same visual representations are used for all motor control tasks. To the best of our knowledge, this is the first self-supervised model to exploit real-world images at scale for motor control. To accelerat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.06173","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-03-11T18:58:10Z","cross_cats_sorted":["cs.LG","cs.RO"],"title_canon_sha256":"f8fcbfabaab281d11ee1891087f70e7141dfac26bd1eec4434568679d9ff2720","abstract_canon_sha256":"4ae18b48233c37bf7045a475623f2a6de15718667c72d2e092eccc938261842c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:13.379655Z","signature_b64":"C9L9ac3HfYjGnKYu9bE6ld1vdUu24Ydlom9fai02cUWbyqLKry2W9yo9fdW5tmwiEOOf1zdI3dhEvpMMqQA6AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e3a7babc953aed6a053ba9adce65697f6794aa9bc6ca98f472122adb3c9ffa71","last_reissued_at":"2026-07-05T04:04:13.379133Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:13.379133Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Masked Visual Pre-training for Motor Control","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Ilija Radosavovic, Jitendra Malik, Tete Xiao, Trevor Darrell","submitted_at":"2022-03-11T18:58:10Z","abstract_excerpt":"This paper shows that self-supervised visual pre-training from real-world images is effective for learning motor control tasks from pixels. We first train the visual representations by masked modeling of natural images. We then freeze the visual encoder and train neural network controllers on top with reinforcement learning. We do not perform any task-specific fine-tuning of the encoder; the same visual representations are used for all motor control tasks. To the best of our knowledge, this is the first self-supervised model to exploit real-world images at scale for motor control. To accelerat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.06173","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.06173/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.06173","created_at":"2026-07-05T04:04:13.379195+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.06173v1","created_at":"2026-07-05T04:04:13.379195+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.06173","created_at":"2026-07-05T04:04:13.379195+00:00"},{"alias_kind":"pith_short_12","alias_value":"4OT3VPEVHLWW","created_at":"2026-07-05T04:04:13.379195+00:00"},{"alias_kind":"pith_short_16","alias_value":"4OT3VPEVHLWWUBJ3","created_at":"2026-07-05T04:04:13.379195+00:00"},{"alias_kind":"pith_short_8","alias_value":"4OT3VPEV","created_at":"2026-07-05T04:04:13.379195+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":30,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06564","citing_title":"Lift3D-VLA: Lifting VLA Models to 3D Geometry and Dynamics-Aware Manipulation","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27325","citing_title":"Not All Actions Are Equal: Rethinking Conditioning for Dexterous World Model","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21100","citing_title":"Factor-Aware Mixture-of-Experts with Pretrained Encoder for Combinatorial Generalization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21406","citing_title":"Robot Self-Improvement via Human-Video Dynamics Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17256","citing_title":"Contrastive Action-Image Pre-training for Visuomotor Control","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17200","citing_title":"ACE-Ego-0: Unifying Egocentric Human and Robotic Data for VLA Pretraining","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17055","citing_title":"T-Rex: Tactile-Reactive Dexterous Manipulation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12499","citing_title":"Action-Effect Memory Pretraining for Robot Manipulation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00054","citing_title":"From Human Videos to Robot Manipulation: A Survey on Scalable Vision-Language-Action Learning with Human-Centric Data","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30350","citing_title":"DynaFLIP: Rethinking Robotics Perception via Tri-Modal-Dynamics Guided Representation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23856","citing_title":"Point Tracking Improves World Action Models","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2504.16054","citing_title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2507.09180","citing_title":"Multimodal Fusion for Sim2real Transfer in Visual Reinforcement Learning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15836","citing_title":"GAP: Geometric Anchor Pre-training for Data-Efficient Visuomotor Learning of Manipulation Tasks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08902","citing_title":"Intention-Conditioned Flow Occupancy Models","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10137","citing_title":"Self-Predictive Representations for Combinatorial Generalization in Behavioral Cloning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12878","citing_title":"Uni-Hand: Universal Hand Motion Forecasting in Egocentric Views","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14427","citing_title":"Self-Supervised Multisensory Pretraining for Contact-Rich Robot Reinforcement Learning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04983","citing_title":"DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15493","citing_title":"GR-3 Technical Report","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2601.07060","citing_title":"PALM: Progress-Aware Policy Learning via Affordance Reasoning for Long-Horizon Robotic Manipulation","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06382","citing_title":"Now You See That: Learning End-to-End Humanoid Locomotion from Raw Pixels","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2508.00795","citing_title":"Video Generators are Robot Policies","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2409.16283","citing_title":"Gen2Act: Human Video Generation in Novel Scenarios enables Generalizable Robot Manipulation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2210.00030","citing_title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5","json":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5.json","graph_json":"https://pith.science/api/pith-number/4OT3VPEVHLWWUBJ3VGW44ZLJP5/graph.json","events_json":"https://pith.science/api/pith-number/4OT3VPEVHLWWUBJ3VGW44ZLJP5/events.json","paper":"https://pith.science/paper/4OT3VPEV"},"agent_actions":{"view_html":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5","download_json":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5.json","view_paper":"https://pith.science/paper/4OT3VPEV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.06173&json=true","fetch_graph":"https://pith.science/api/pith-number/4OT3VPEVHLWWUBJ3VGW44ZLJP5/graph.json","fetch_events":"https://pith.science/api/pith-number/4OT3VPEVHLWWUBJ3VGW44ZLJP5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5/action/storage_attestation","attest_author":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5/action/author_attestation","sign_citation":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5/action/citation_signature","submit_replication":"https://pith.science/pith/4OT3VPEVHLWWUBJ3VGW44ZLJP5/action/replication_record"}},"created_at":"2026-07-05T04:04:13.379195+00:00","updated_at":"2026-07-05T04:04:13.379195+00:00"}