{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:U324ONUSO3HWI3TGKDVY66LFH7","short_pith_number":"pith:U324ONUS","schema_version":"1.0","canonical_sha256":"a6f5c7369276cf646e6650eb8f79653fdc8a844fb738d7e0e118034bc5600b75","source":{"kind":"arxiv","id":"2407.04152","version":2},"attestation_state":"computed","paper":{"title":"VoxAct-B: Voxel-Based Acting and Stabilizing Policy for Bimanual Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Daniel Seita, Gaurav Sukhatme, I-Chun Arthur Liu, Sicheng He","submitted_at":"2024-07-04T20:58:20Z","abstract_excerpt":"Bimanual manipulation is critical to many robotics applications. In contrast to single-arm manipulation, bimanual manipulation tasks are challenging due to higher-dimensional action spaces. Prior works leverage large amounts of data and primitive actions to address this problem, but may suffer from sample inefficiency and limited generalization across various tasks. To this end, we propose VoxAct-B, a language-conditioned, voxel-based method that leverages Vision Language Models (VLMs) to prioritize key regions within the scene and reconstruct a voxel grid. We provide this voxel grid to our bi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.04152","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-07-04T20:58:20Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"5e0c5d36ec974d409bd2644c38dbf9392e34725cd14476f21381ebf46d60497a","abstract_canon_sha256":"8ac28072df212007c64bc00ee28c5c896e2bde5d07bae9d3ad600a40076d760e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:19.369616Z","signature_b64":"coClOhs6TrSQj1qs9OpRWAisyPfwmbn0D7WNp+/XkqjNmCAnGAcxZcFGrjtIukFN58UpUh/zoOKEdfsQbA3OAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6f5c7369276cf646e6650eb8f79653fdc8a844fb738d7e0e118034bc5600b75","last_reissued_at":"2026-07-05T09:16:19.369125Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:19.369125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VoxAct-B: Voxel-Based Acting and Stabilizing Policy for Bimanual Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Daniel Seita, Gaurav Sukhatme, I-Chun Arthur Liu, Sicheng He","submitted_at":"2024-07-04T20:58:20Z","abstract_excerpt":"Bimanual manipulation is critical to many robotics applications. In contrast to single-arm manipulation, bimanual manipulation tasks are challenging due to higher-dimensional action spaces. Prior works leverage large amounts of data and primitive actions to address this problem, but may suffer from sample inefficiency and limited generalization across various tasks. To this end, we propose VoxAct-B, a language-conditioned, voxel-based method that leverages Vision Language Models (VLMs) to prioritize key regions within the scene and reconstruct a voxel grid. We provide this voxel grid to our bi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.04152","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.04152/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.04152","created_at":"2026-07-05T09:16:19.369182+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.04152v2","created_at":"2026-07-05T09:16:19.369182+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.04152","created_at":"2026-07-05T09:16:19.369182+00:00"},{"alias_kind":"pith_short_12","alias_value":"U324ONUSO3HW","created_at":"2026-07-05T09:16:19.369182+00:00"},{"alias_kind":"pith_short_16","alias_value":"U324ONUSO3HWI3TG","created_at":"2026-07-05T09:16:19.369182+00:00"},{"alias_kind":"pith_short_8","alias_value":"U324ONUS","created_at":"2026-07-05T09:16:19.369182+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13452","citing_title":"CUBic: Coordinated Unified Bimanual Perception and Control Framework","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19522","citing_title":"GenerativeMPC: VLM-RAG-guided Whole-Body MPC with Virtual Impedance for Bimanual Mobile Manipulation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2410.07864","citing_title":"RDT-1B: a Diffusion Foundation Model for Bimanual Manipulation","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7","json":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7.json","graph_json":"https://pith.science/api/pith-number/U324ONUSO3HWI3TGKDVY66LFH7/graph.json","events_json":"https://pith.science/api/pith-number/U324ONUSO3HWI3TGKDVY66LFH7/events.json","paper":"https://pith.science/paper/U324ONUS"},"agent_actions":{"view_html":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7","download_json":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7.json","view_paper":"https://pith.science/paper/U324ONUS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.04152&json=true","fetch_graph":"https://pith.science/api/pith-number/U324ONUSO3HWI3TGKDVY66LFH7/graph.json","fetch_events":"https://pith.science/api/pith-number/U324ONUSO3HWI3TGKDVY66LFH7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7/action/storage_attestation","attest_author":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7/action/author_attestation","sign_citation":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7/action/citation_signature","submit_replication":"https://pith.science/pith/U324ONUSO3HWI3TGKDVY66LFH7/action/replication_record"}},"created_at":"2026-07-05T09:16:19.369182+00:00","updated_at":"2026-07-05T09:16:19.369182+00:00"}