{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:G2WY72TJKRT6GQCTU3ZYV2VARG","short_pith_number":"pith:G2WY72TJ","schema_version":"1.0","canonical_sha256":"36ad8fea695467e34053a6f38aeaa089b3d2322ec3a890db72a9f83b26dbd084","source":{"kind":"arxiv","id":"2603.09971","version":2},"attestation_state":"computed","paper":{"title":"TiPToP: A Modular Open-Vocabulary Robot Manipulation System That Plans","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Christopher Watson, Dinesh Jayaraman, Edward Hu, Jie Wang, Jing Cao, Leslie Pack Kaelbling, Nishanth Kumar, Ryan Lindeborg, Sahit Chintalapudi, Tom\\'as Lozano-P\\'erez, William Shen","submitted_at":"2026-03-10T17:59:00Z","abstract_excerpt":"We present TiPToP, a modular manipulation system that integrates pretrained foundation models with a GPU-accelerated Task and Motion Planner to solve tasks directly from RGB images and natural language. TiPToP composes perception, planning, and execution modules and requires no robot training data. It can be deployed on a standard DROID setup in under an hour and adapted to new embodiments with minimal effort. We evaluate TiPToP against $\\pi_{0.5}\\text{-DROID}$, a state-of-the-art VLA fine-tuned on 350 hours of demonstrations, across two real-world DROID setups (one operated by an external tea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.09971","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2026-03-10T17:59:00Z","cross_cats_sorted":[],"title_canon_sha256":"3f6665ff0ed092365957a862b3d71d1b83700c07ab07a2304a2e98c435fe78c5","abstract_canon_sha256":"c3c3c7ee643a8307335b72bdd0bfffa2148967fadadca2f59834411a1da437bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36ad8fea695467e34053a6f38aeaa089b3d2322ec3a890db72a9f83b26dbd084","last_reissued_at":"2026-07-30T00:08:13.488952Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-30T00:08:13.488952Z"},"graph_snapshot":{"paper":{"title":"TiPToP: A Modular Open-Vocabulary Robot Manipulation System That Plans","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Christopher Watson, Dinesh Jayaraman, Edward Hu, Jie Wang, Jing Cao, Leslie Pack Kaelbling, Nishanth Kumar, Ryan Lindeborg, Sahit Chintalapudi, Tom\\'as Lozano-P\\'erez, William Shen","submitted_at":"2026-03-10T17:59:00Z","abstract_excerpt":"We present TiPToP, a modular manipulation system that integrates pretrained foundation models with a GPU-accelerated Task and Motion Planner to solve tasks directly from RGB images and natural language. TiPToP composes perception, planning, and execution modules and requires no robot training data. It can be deployed on a standard DROID setup in under an hour and adapted to new embodiments with minimal effort. We evaluate TiPToP against $\\pi_{0.5}\\text{-DROID}$, a state-of-the-art VLA fine-tuned on 350 hours of demonstrations, across two real-world DROID setups (one operated by an external tea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.09971","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.09971/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.09971","created_at":"2026-07-30T00:08:13.492543+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.09971v2","created_at":"2026-07-30T00:08:13.492543+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.09971","created_at":"2026-07-30T00:08:13.492543+00:00"},{"alias_kind":"pith_short_12","alias_value":"G2WY72TJKRT6","created_at":"2026-07-30T00:08:13.492543+00:00"},{"alias_kind":"pith_short_16","alias_value":"G2WY72TJKRT6GQCT","created_at":"2026-07-30T00:08:13.492543+00:00"},{"alias_kind":"pith_short_8","alias_value":"G2WY72TJ","created_at":"2026-07-30T00:08:13.492543+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":6,"sample":[{"citing_arxiv_id":"2607.06501","citing_title":"Hypothesis-driven Model Expansion under Uncertainty for Open-World Robot Planning","ref_index":52,"is_internal_anchor":true},{"citing_arxiv_id":"2606.13675","citing_title":"Improving Robotic Generalist Policies via Flow Reversal Steering","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07723","citing_title":"VoLo: A Physical Orchestrator for Open-Vocabulary Long-Horizon Manipulation","ref_index":61,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06061","citing_title":"A Conversational Framework for Human-Robot Collaborative Manipulation with Distributed Generative AI models","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2606.00998","citing_title":"GraspGen-X: Cross-Embodiment 6-DOF Diffusion-based Grasping","ref_index":56,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15975","citing_title":"Learning Bilevel Policies over Symbolic World Models for Long-Horizon Planning","ref_index":116,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG","json":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG.json","graph_json":"https://pith.science/api/pith-number/G2WY72TJKRT6GQCTU3ZYV2VARG/graph.json","events_json":"https://pith.science/api/pith-number/G2WY72TJKRT6GQCTU3ZYV2VARG/events.json","paper":"https://pith.science/paper/G2WY72TJ"},"agent_actions":{"view_html":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG","download_json":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG.json","view_paper":"https://pith.science/paper/G2WY72TJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.09971&json=true","fetch_graph":"https://pith.science/api/pith-number/G2WY72TJKRT6GQCTU3ZYV2VARG/graph.json","fetch_events":"https://pith.science/api/pith-number/G2WY72TJKRT6GQCTU3ZYV2VARG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG/action/storage_attestation","attest_author":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG/action/author_attestation","sign_citation":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG/action/citation_signature","submit_replication":"https://pith.science/pith/G2WY72TJKRT6GQCTU3ZYV2VARG/action/replication_record"}},"created_at":"2026-07-30T00:08:13.492543+00:00","updated_at":"2026-07-30T00:08:13.492543+00:00"}