{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:A335MGICPKJSB2ACNFMSI4LUBI","short_pith_number":"pith:A335MGIC","schema_version":"1.0","canonical_sha256":"06f7d619027a9320e80269592471740a181a53a764bba451eeb7ed7085d1ecbb","source":{"kind":"arxiv","id":"2212.05711","version":2},"attestation_state":"computed","paper":{"title":"CACTI: A Framework for Scalable Multi-Task Multi-Scene Visual Imitation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Aravind Rajeswaran, Homanga Bharadhwaj, Shuran Song, Vikash Kumar, Vincent Moens, Zhao Mandi","submitted_at":"2022-12-12T05:30:08Z","abstract_excerpt":"Large-scale training have propelled significant progress in various sub-fields of AI such as computer vision and natural language processing. However, building robot learning systems at a comparable scale remains challenging. To develop robots that can perform a wide range of skills and adapt to new scenarios, efficient methods for collecting vast and diverse amounts of data on physical robot systems are required, as well as the capability to train high-capacity policies using such datasets. In this work, we propose a framework for scaling robot learning, with specific focus on multi-task and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.05711","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2022-12-12T05:30:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8661c538f7a6721cb9fd030be1162a50be8f0bd69fda2f695f6a4794f36fcee0","abstract_canon_sha256":"de29c4b86907702add8046f5c78e4423c71978c7707170dabc424013f3ef4732"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:42:29.656173Z","signature_b64":"yb14v6i53+44KBSZNaXknY6kVtdUpHOEBPhe27XFmlKYk5myS6IqGwF7ImlsEZ4kip77kxLs0aA8PrA3Og48Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06f7d619027a9320e80269592471740a181a53a764bba451eeb7ed7085d1ecbb","last_reissued_at":"2026-07-05T05:42:29.655648Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:42:29.655648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CACTI: A Framework for Scalable Multi-Task Multi-Scene Visual Imitation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Aravind Rajeswaran, Homanga Bharadhwaj, Shuran Song, Vikash Kumar, Vincent Moens, Zhao Mandi","submitted_at":"2022-12-12T05:30:08Z","abstract_excerpt":"Large-scale training have propelled significant progress in various sub-fields of AI such as computer vision and natural language processing. However, building robot learning systems at a comparable scale remains challenging. To develop robots that can perform a wide range of skills and adapt to new scenarios, efficient methods for collecting vast and diverse amounts of data on physical robot systems are required, as well as the capability to train high-capacity policies using such datasets. In this work, we propose a framework for scaling robot learning, with specific focus on multi-task and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.05711","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.05711/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.05711","created_at":"2026-07-05T05:42:29.655728+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.05711v2","created_at":"2026-07-05T05:42:29.655728+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.05711","created_at":"2026-07-05T05:42:29.655728+00:00"},{"alias_kind":"pith_short_12","alias_value":"A335MGICPKJS","created_at":"2026-07-05T05:42:29.655728+00:00"},{"alias_kind":"pith_short_16","alias_value":"A335MGICPKJSB2AC","created_at":"2026-07-05T05:42:29.655728+00:00"},{"alias_kind":"pith_short_8","alias_value":"A335MGIC","created_at":"2026-07-05T05:42:29.655728+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10918","citing_title":"Task Robustness via Re-Labelling Vision-Action Robot Data","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2302.11550","citing_title":"Scaling Robot Learning with Semantically Imagined Experience","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2310.17596","citing_title":"MimicGen: A Data Generation System for Scalable Robot Learning using Human Demonstrations","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2310.10639","citing_title":"Zero-Shot Robotic Manipulation with Pretrained Image-Editing Diffusion Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12705","citing_title":"DreamGen: Unlocking Generalization in Robot Learning through Video World Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2409.16283","citing_title":"Gen2Act: Human Video Generation in Novel Scenarios enables Generalizable Robot Manipulation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17177","citing_title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2307.05973","citing_title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00244","citing_title":"Lucid-XR: An Extended-Reality Data Engine for Robotic Manipulation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14734","citing_title":"GR00T N1: An Open Foundation Model for Generalist Humanoid Robots","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI","json":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI.json","graph_json":"https://pith.science/api/pith-number/A335MGICPKJSB2ACNFMSI4LUBI/graph.json","events_json":"https://pith.science/api/pith-number/A335MGICPKJSB2ACNFMSI4LUBI/events.json","paper":"https://pith.science/paper/A335MGIC"},"agent_actions":{"view_html":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI","download_json":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI.json","view_paper":"https://pith.science/paper/A335MGIC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.05711&json=true","fetch_graph":"https://pith.science/api/pith-number/A335MGICPKJSB2ACNFMSI4LUBI/graph.json","fetch_events":"https://pith.science/api/pith-number/A335MGICPKJSB2ACNFMSI4LUBI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI/action/storage_attestation","attest_author":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI/action/author_attestation","sign_citation":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI/action/citation_signature","submit_replication":"https://pith.science/pith/A335MGICPKJSB2ACNFMSI4LUBI/action/replication_record"}},"created_at":"2026-07-05T05:42:29.655728+00:00","updated_at":"2026-07-05T05:42:29.655728+00:00"}