{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NZ6WGPCPNIN26PXAW7WDHXJSQ4","short_pith_number":"pith:NZ6WGPCP","schema_version":"1.0","canonical_sha256":"6e7d633c4f6a1baf3ee0b7ec33dd328712e2acf8b39ee410fb8e21e040069e07","source":{"kind":"arxiv","id":"2309.14320","version":1},"attestation_state":"computed","paper":{"title":"MUTEX: Learning Unified Policies from Multimodal Task Specifications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Roberto Mart\\'in-Mart\\'in, Rutav Shah, Yuke Zhu","submitted_at":"2023-09-25T17:45:31Z","abstract_excerpt":"Humans use different modalities, such as speech, text, images, videos, etc., to communicate their intent and goals with teammates. For robots to become better assistants, we aim to endow them with the ability to follow instructions and understand tasks specified by their human partners. Most robotic policy learning methods have focused on one single modality of task specification while ignoring the rich cross-modal information. We present MUTEX, a unified approach to policy learning from multimodal task specifications. It trains a transformer-based architecture to facilitate cross-modal reason"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.14320","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-09-25T17:45:31Z","cross_cats_sorted":[],"title_canon_sha256":"92a9394d97721a45503c885a565aa2aa7339c261722d702cbf83dedf591e5be8","abstract_canon_sha256":"790a10d85bed776cf1f1f4a3212f0ab9b1590fd86a8aae58acd7249f5d48f1f4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:54:09.563290Z","signature_b64":"z89LG8Hq0yr2RqX35HLiZ6SqB9AsGF8nVHCAnV9bbsZpBcqwrgylaWyfv4qldnZwfiS3+s9ymOrTvHhemm9TAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e7d633c4f6a1baf3ee0b7ec33dd328712e2acf8b39ee410fb8e21e040069e07","last_reissued_at":"2026-07-05T06:54:09.562813Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:54:09.562813Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MUTEX: Learning Unified Policies from Multimodal Task Specifications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Roberto Mart\\'in-Mart\\'in, Rutav Shah, Yuke Zhu","submitted_at":"2023-09-25T17:45:31Z","abstract_excerpt":"Humans use different modalities, such as speech, text, images, videos, etc., to communicate their intent and goals with teammates. For robots to become better assistants, we aim to endow them with the ability to follow instructions and understand tasks specified by their human partners. Most robotic policy learning methods have focused on one single modality of task specification while ignoring the rich cross-modal information. We present MUTEX, a unified approach to policy learning from multimodal task specifications. It trains a transformer-based architecture to facilitate cross-modal reason"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.14320","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.14320/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.14320","created_at":"2026-07-05T06:54:09.562872+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.14320v1","created_at":"2026-07-05T06:54:09.562872+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.14320","created_at":"2026-07-05T06:54:09.562872+00:00"},{"alias_kind":"pith_short_12","alias_value":"NZ6WGPCPNIN2","created_at":"2026-07-05T06:54:09.562872+00:00"},{"alias_kind":"pith_short_16","alias_value":"NZ6WGPCPNIN26PXA","created_at":"2026-07-05T06:54:09.562872+00:00"},{"alias_kind":"pith_short_8","alias_value":"NZ6WGPCP","created_at":"2026-07-05T06:54:09.562872+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30011","citing_title":"VisualThink-VLA: Visual Intermediate Reasoning for Effective and Low-Latency Vision-Language-Action Policies","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2401.03568","citing_title":"Agent AI: Surveying the Horizons of Multimodal Interaction","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13403","citing_title":"RotVLA: Rotational Latent Action for Vision-Language-Action Model","ref_index":69,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4","json":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4.json","graph_json":"https://pith.science/api/pith-number/NZ6WGPCPNIN26PXAW7WDHXJSQ4/graph.json","events_json":"https://pith.science/api/pith-number/NZ6WGPCPNIN26PXAW7WDHXJSQ4/events.json","paper":"https://pith.science/paper/NZ6WGPCP"},"agent_actions":{"view_html":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4","download_json":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4.json","view_paper":"https://pith.science/paper/NZ6WGPCP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.14320&json=true","fetch_graph":"https://pith.science/api/pith-number/NZ6WGPCPNIN26PXAW7WDHXJSQ4/graph.json","fetch_events":"https://pith.science/api/pith-number/NZ6WGPCPNIN26PXAW7WDHXJSQ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4/action/storage_attestation","attest_author":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4/action/author_attestation","sign_citation":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4/action/citation_signature","submit_replication":"https://pith.science/pith/NZ6WGPCPNIN26PXAW7WDHXJSQ4/action/replication_record"}},"created_at":"2026-07-05T06:54:09.562872+00:00","updated_at":"2026-07-05T06:54:09.562872+00:00"}