{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:43QNJ7CL2HBQOJNG6JO5H74FBQ","short_pith_number":"pith:43QNJ7CL","schema_version":"1.0","canonical_sha256":"e6e0d4fc4bd1c30725a6f25dd3ff850c03f01b3cdfa416912225827faac9e83f","source":{"kind":"arxiv","id":"2401.12202","version":2},"attestation_state":"computed","paper":{"title":"OK-Robot: What Really Matters in Integrating Open-Knowledge Models for Robotics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Chris Paxton, Jay Vakil, Lerrel Pinto, Nur Muhammad Mahi Shafiullah, Peiqi Liu, Yaswanth Orru","submitted_at":"2024-01-22T18:42:20Z","abstract_excerpt":"Remarkable progress has been made in recent years in the fields of vision, language, and robotics. We now have vision models capable of recognizing objects based on language queries, navigation systems that can effectively control mobile systems, and grasping models that can handle a wide range of objects. Despite these advancements, general-purpose applications of robotics still lag behind, even though they rely on these fundamental capabilities of recognition, navigation, and grasping. In this paper, we adopt a systems-first approach to develop a new Open Knowledge-based robotics framework c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.12202","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-01-22T18:42:20Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"6d4a3ed24e77287f4c7757055900cbff683476d7d99e4d88614049e466812e16","abstract_canon_sha256":"9a3808101695fe9b129c2c8a51aa2a3da2c0d9c68901daaca79458dbf499d4c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:37:34.646429Z","signature_b64":"769tpw+H5b89P3Fn0IwwMEFSlRsMv45VV81EG6RHq2Hjlf6UISUB3P0R6Gt/xSf4Ce3749rS3aLdH9u4c/gQCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6e0d4fc4bd1c30725a6f25dd3ff850c03f01b3cdfa416912225827faac9e83f","last_reissued_at":"2026-07-05T09:37:34.645900Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:37:34.645900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OK-Robot: What Really Matters in Integrating Open-Knowledge Models for Robotics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Chris Paxton, Jay Vakil, Lerrel Pinto, Nur Muhammad Mahi Shafiullah, Peiqi Liu, Yaswanth Orru","submitted_at":"2024-01-22T18:42:20Z","abstract_excerpt":"Remarkable progress has been made in recent years in the fields of vision, language, and robotics. We now have vision models capable of recognizing objects based on language queries, navigation systems that can effectively control mobile systems, and grasping models that can handle a wide range of objects. Despite these advancements, general-purpose applications of robotics still lag behind, even though they rely on these fundamental capabilities of recognition, navigation, and grasping. In this paper, we adopt a systems-first approach to develop a new Open Knowledge-based robotics framework c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.12202","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.12202/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.12202","created_at":"2026-07-05T09:37:34.645958+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.12202v2","created_at":"2026-07-05T09:37:34.645958+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.12202","created_at":"2026-07-05T09:37:34.645958+00:00"},{"alias_kind":"pith_short_12","alias_value":"43QNJ7CL2HBQ","created_at":"2026-07-05T09:37:34.645958+00:00"},{"alias_kind":"pith_short_16","alias_value":"43QNJ7CL2HBQOJNG","created_at":"2026-07-05T09:37:34.645958+00:00"},{"alias_kind":"pith_short_8","alias_value":"43QNJ7CL","created_at":"2026-07-05T09:37:34.645958+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25136","citing_title":"Memory Retrieval in Visuomotor Policies for Long-Horizon Robot Control","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23565","citing_title":"HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18646","citing_title":"A Scalable Embodied Intelligence Platform for Seamless Real-to-Sim-to-Real Transfer of Household Mobile Manipulation Tasks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13394","citing_title":"GeoHAT: Geometry-Adaptive Hybrid Action Transformer for Mobile Manipulation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03374","citing_title":"eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00998","citing_title":"GraspGen-X: Cross-Embodiment 6-DOF Diffusion-based Grasping","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00985","citing_title":"Make Your VLA More Robust Without More Data By Interleaving Motion Planning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2409.13107","citing_title":"Towards Robust Surgical Automation via Digital Twin Representations from Foundation Models","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06239","citing_title":"Open-Architecture End-to-End System for Real-World Autonomous Robot Navigation","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2504.16054","citing_title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06224","citing_title":"Uni-NaVid: A Video-based Vision-Language-Action Model for Unifying Embodied Navigation Tasks","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2502.19417","citing_title":"Hi Robot: Open-Ended Instruction Following with Hierarchical Vision-Language-Action Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28197","citing_title":"OmniRobotHome: A Multi-Camera Platform for Real-Time Multiadic Human-Robot Interaction","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02487","citing_title":"Visibility-Aware Mobile Grasping in Dynamic Environments","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2410.24164","citing_title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02487","citing_title":"Visibility-Aware Mobile Grasping in Dynamic Environments","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ","json":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ.json","graph_json":"https://pith.science/api/pith-number/43QNJ7CL2HBQOJNG6JO5H74FBQ/graph.json","events_json":"https://pith.science/api/pith-number/43QNJ7CL2HBQOJNG6JO5H74FBQ/events.json","paper":"https://pith.science/paper/43QNJ7CL"},"agent_actions":{"view_html":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ","download_json":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ.json","view_paper":"https://pith.science/paper/43QNJ7CL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.12202&json=true","fetch_graph":"https://pith.science/api/pith-number/43QNJ7CL2HBQOJNG6JO5H74FBQ/graph.json","fetch_events":"https://pith.science/api/pith-number/43QNJ7CL2HBQOJNG6JO5H74FBQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ/action/storage_attestation","attest_author":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ/action/author_attestation","sign_citation":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ/action/citation_signature","submit_replication":"https://pith.science/pith/43QNJ7CL2HBQOJNG6JO5H74FBQ/action/replication_record"}},"created_at":"2026-07-05T09:37:34.645958+00:00","updated_at":"2026-07-05T09:37:34.645958+00:00"}