{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BVQLIHY5YUYNSRZ6U6ZDHMJSZD","short_pith_number":"pith:BVQLIHY5","schema_version":"1.0","canonical_sha256":"0d60b41f1dc530d9473ea7b233b132c8f88a44fcae226fa6176830f144834ba3","source":{"kind":"arxiv","id":"2407.20179","version":2},"attestation_state":"computed","paper":{"title":"Theia: Distilling Diverse Vision Foundation Models for Robot Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Brandon B. May, David Watkins, Jinghuan Shang, Karl Schmeckpeper, Laura Herlant, Maria Vittoria Minniti, Tarik Kelestemur","submitted_at":"2024-07-29T17:08:21Z","abstract_excerpt":"Vision-based robot policy learning, which maps visual inputs to actions, necessitates a holistic understanding of diverse visual tasks beyond single-task needs like classification or segmentation. Inspired by this, we introduce Theia, a vision foundation model for robot learning that distills multiple off-the-shelf vision foundation models trained on varied vision tasks. Theia's rich visual representations encode diverse visual knowledge, enhancing downstream robot learning. Extensive experiments demonstrate that Theia outperforms its teacher models and prior robot learning models using less t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20179","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-07-29T17:08:21Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"5faec412b78e6c5e9b11bb96b0ff53392bf9841e93f290ad8b18d1b41f32443d","abstract_canon_sha256":"76cf14d164b6ec700a3d47f45b75a70508f549269d887774ffd1f9dbd6138f50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:28.964045Z","signature_b64":"FvMFu4ep0SmHXQwy3yHg4a0iWPa9K1hZvb05NmcwbzDJNwwFcmWTMbvc7ETXGSUiAo8qvp4LMxWFQri0Irp3Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d60b41f1dc530d9473ea7b233b132c8f88a44fcae226fa6176830f144834ba3","last_reissued_at":"2026-07-05T09:18:28.963554Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:28.963554Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Theia: Distilling Diverse Vision Foundation Models for Robot Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Brandon B. May, David Watkins, Jinghuan Shang, Karl Schmeckpeper, Laura Herlant, Maria Vittoria Minniti, Tarik Kelestemur","submitted_at":"2024-07-29T17:08:21Z","abstract_excerpt":"Vision-based robot policy learning, which maps visual inputs to actions, necessitates a holistic understanding of diverse visual tasks beyond single-task needs like classification or segmentation. Inspired by this, we introduce Theia, a vision foundation model for robot learning that distills multiple off-the-shelf vision foundation models trained on varied vision tasks. Theia's rich visual representations encode diverse visual knowledge, enhancing downstream robot learning. Extensive experiments demonstrate that Theia outperforms its teacher models and prior robot learning models using less t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20179","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20179/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20179","created_at":"2026-07-05T09:18:28.963609+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20179v2","created_at":"2026-07-05T09:18:28.963609+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20179","created_at":"2026-07-05T09:18:28.963609+00:00"},{"alias_kind":"pith_short_12","alias_value":"BVQLIHY5YUYN","created_at":"2026-07-05T09:18:28.963609+00:00"},{"alias_kind":"pith_short_16","alias_value":"BVQLIHY5YUYNSRZ6","created_at":"2026-07-05T09:18:28.963609+00:00"},{"alias_kind":"pith_short_8","alias_value":"BVQLIHY5","created_at":"2026-07-05T09:18:28.963609+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20559","citing_title":"UNIEGO: Proxies as Mediators for Unified Egocentric Video Representation Learning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12499","citing_title":"Action-Effect Memory Pretraining for Robot Manipulation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04046","citing_title":"Dive into the Scene: Breaking the Perceptual Bottleneck in Vision-Language Decision Making via Focus Plan Generation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03444","citing_title":"PRISM: Synergizing Vision Foundation Models via Self-organized Expert Specialization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16743","citing_title":"LACE: Latent Visual Representation for Cross-Embodiment Learning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19704","citing_title":"RADSeg: Unleashing Parameter and Compute Efficient Zero-Shot Open-Vocabulary Segmentation Using Agglomerative Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20157","citing_title":"SigLino: Efficient Multi-Teacher Distillation for Agglomerative Vision Foundation Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04447","citing_title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15956","citing_title":"ExpertGen: Scalable Sim-to-Real Expert Policy Learning from Imperfect Behavior Priors","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12491","citing_title":"Elastic Attention Cores for Scalable Vision Transformers","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22992","citing_title":"Efficient Image Annotation via Semi-Supervised Object Segmentation with Label Propagation","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD","json":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD.json","graph_json":"https://pith.science/api/pith-number/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/graph.json","events_json":"https://pith.science/api/pith-number/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/events.json","paper":"https://pith.science/paper/BVQLIHY5"},"agent_actions":{"view_html":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD","download_json":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD.json","view_paper":"https://pith.science/paper/BVQLIHY5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20179&json=true","fetch_graph":"https://pith.science/api/pith-number/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/graph.json","fetch_events":"https://pith.science/api/pith-number/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/action/storage_attestation","attest_author":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/action/author_attestation","sign_citation":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/action/citation_signature","submit_replication":"https://pith.science/pith/BVQLIHY5YUYNSRZ6U6ZDHMJSZD/action/replication_record"}},"created_at":"2026-07-05T09:18:28.963609+00:00","updated_at":"2026-07-05T09:18:28.963609+00:00"}