{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ARRA5IFI3XG5WB74A5VBZLESW2","short_pith_number":"pith:ARRA5IFI","schema_version":"1.0","canonical_sha256":"04620ea0a8ddcddb07fc076a1cac92b6af252131450711d80b55cccffb909c6e","source":{"kind":"arxiv","id":"2406.08545","version":1},"attestation_state":"computed","paper":{"title":"RVT-2: Learning Precise Manipulation from Few Demonstrations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Ankit Goyal, Dieter Fox, Jie Xu, Valts Blukis, Yijie Guo, Yu-Wei Chao","submitted_at":"2024-06-12T18:00:01Z","abstract_excerpt":"In this work, we study how to build a robotic system that can solve multiple 3D manipulation tasks given language instructions. To be useful in industrial and household domains, such a system should be capable of learning new tasks with few demonstrations and solving them precisely. Prior works, like PerAct and RVT, have studied this problem, however, they often struggle with tasks requiring high precision. We study how to make them more effective, precise, and fast. Using a combination of architectural and system-level improvements, we propose RVT-2, a multitask 3D manipulation model that is "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08545","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-06-12T18:00:01Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"61e22f28f3b8ae264c79098b144080e0b23e0511d339479977733f868e260871","abstract_canon_sha256":"bfea98861486f6e911f42c4ff21453f11d02ce270924cfd40f42ce14f3fc8616"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:18.720457Z","signature_b64":"usZAuMjLM6YKB82de54DNc4t+iMskK025kY4JT846IXrcW+3mH8o8RlZzG61oV5lYTIQ3OMuwAufq9uuT1CGDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04620ea0a8ddcddb07fc076a1cac92b6af252131450711d80b55cccffb909c6e","last_reissued_at":"2026-07-05T08:31:18.719885Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:18.719885Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RVT-2: Learning Precise Manipulation from Few Demonstrations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Ankit Goyal, Dieter Fox, Jie Xu, Valts Blukis, Yijie Guo, Yu-Wei Chao","submitted_at":"2024-06-12T18:00:01Z","abstract_excerpt":"In this work, we study how to build a robotic system that can solve multiple 3D manipulation tasks given language instructions. To be useful in industrial and household domains, such a system should be capable of learning new tasks with few demonstrations and solving them precisely. Prior works, like PerAct and RVT, have studied this problem, however, they often struggle with tasks requiring high precision. We study how to make them more effective, precise, and fast. Using a combination of architectural and system-level improvements, we propose RVT-2, a multitask 3D manipulation model that is "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08545","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08545/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08545","created_at":"2026-07-05T08:31:18.719967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08545v1","created_at":"2026-07-05T08:31:18.719967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08545","created_at":"2026-07-05T08:31:18.719967+00:00"},{"alias_kind":"pith_short_12","alias_value":"ARRA5IFI3XG5","created_at":"2026-07-05T08:31:18.719967+00:00"},{"alias_kind":"pith_short_16","alias_value":"ARRA5IFI3XG5WB74","created_at":"2026-07-05T08:31:18.719967+00:00"},{"alias_kind":"pith_short_8","alias_value":"ARRA5IFI","created_at":"2026-07-05T08:31:18.719967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23625","citing_title":"Learning to See While Learning to Act: Diffusion Models for Active Perception in Robot Imitation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21501","citing_title":"UniviewVLA: A Unified Multiview Vision-Language-Action Model with World Modeling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20867","citing_title":"FOCA: Future-Oriented Conditioning for Data-Efficient Vision-Language-Action Adaptation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13394","citing_title":"GeoHAT: Geometry-Adaptive Hybrid Action Transformer for Mobile Manipulation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13578","citing_title":"LabVLA: Grounding Vision-Language-Action Models in Scientific Laboratories","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01072","citing_title":"Expanding Spatial and Temporal Context for Robotic Imitation Learning With Scene Graphs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24642","citing_title":"Understanding the Impact of Geometric Foundation Models on Vision-Language-Action Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25829","citing_title":"OASIS: Observation-Action Space Alignment via SE(3) Trajectory Prediction for Robotic Manipulation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14093","citing_title":"A Survey on Vision-Language-Action Models for Embodied AI","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21862","citing_title":"EvoScene-VLA: Evolving Scene Beliefs Inside the Action Decoder for Chunked Robot Control","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21258","citing_title":"Learning Structural Latent Points for Efficient Visual Representations in Robotic Manipulation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2505.03233","citing_title":"GraspVLA: a Grasping Foundation Model Pre-trained on Billion-scale Synthetic Action Data","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2409.01652","citing_title":"ReKep: Spatio-Temporal Reasoning of Relational Keypoint Constraints for Robotic Manipulation","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2503.10631","citing_title":"HybridVLA: Collaborative Diffusion and Autoregression in a Unified Vision-Language-Action Model","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05117","citing_title":"SeedPolicy: Horizon Scaling via Self-Evolving Diffusion Policy for Robot Manipulation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01448","citing_title":"Decompose and Recompose: Reasoning New Skills from Existing Abilities for Cross-Task Robotic Manipulation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10573","citing_title":"Learning 3D Representations for Spatial Intelligence from Unposed Multi-View Images","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05672","citing_title":"A1: A Fully Transparent Open-Source, Adaptive and Efficient Truncated Vision-Language-Action Model","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2","json":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2.json","graph_json":"https://pith.science/api/pith-number/ARRA5IFI3XG5WB74A5VBZLESW2/graph.json","events_json":"https://pith.science/api/pith-number/ARRA5IFI3XG5WB74A5VBZLESW2/events.json","paper":"https://pith.science/paper/ARRA5IFI"},"agent_actions":{"view_html":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2","download_json":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2.json","view_paper":"https://pith.science/paper/ARRA5IFI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08545&json=true","fetch_graph":"https://pith.science/api/pith-number/ARRA5IFI3XG5WB74A5VBZLESW2/graph.json","fetch_events":"https://pith.science/api/pith-number/ARRA5IFI3XG5WB74A5VBZLESW2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2/action/storage_attestation","attest_author":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2/action/author_attestation","sign_citation":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2/action/citation_signature","submit_replication":"https://pith.science/pith/ARRA5IFI3XG5WB74A5VBZLESW2/action/replication_record"}},"created_at":"2026-07-05T08:31:18.719967+00:00","updated_at":"2026-07-05T08:31:18.719967+00:00"}