{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QEL74OMROBNICIVL5JAM3D2ZHN","short_pith_number":"pith:QEL74OMR","schema_version":"1.0","canonical_sha256":"8117fe3991705a8122abea40cd8f593b6ad118161a0c9fa6a0b77d165762e2e0","source":{"kind":"arxiv","id":"2509.07962","version":1},"attestation_state":"computed","paper":{"title":"TA-VLA: Elucidating the Design Space of Torque-aware Vision-Language-Action Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chenghao Yue, Haobo Xu, Hao Zhao, Huan-ang Gao, Zehao Lin, Zhuo Yang, Ziwei Wang, Zongzheng Zhang","submitted_at":"2025-09-09T17:50:37Z","abstract_excerpt":"Many robotic manipulation tasks require sensing and responding to force signals such as torque to assess whether the task has been successfully completed and to enable closed-loop control. However, current Vision-Language-Action (VLA) models lack the ability to integrate such subtle physical feedback. In this work, we explore Torque-aware VLA models, aiming to bridge this gap by systematically studying the design space for incorporating torque signals into existing VLA architectures. We identify and evaluate several strategies, leading to three key findings. First, introducing torque adapters "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.07962","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-09-09T17:50:37Z","cross_cats_sorted":[],"title_canon_sha256":"0dd48f2776f14dbddc1348cc6aca6c542c43a935024a7ae0a804c94e16a080c2","abstract_canon_sha256":"610b9db7eef1538f17d4284f9a4249e5f879a38b414b2ce456fead9678dd11e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:43.105954Z","signature_b64":"fu7T1c5P91qRd1kbbONxAv+9CyMD2aT4UZWh8p+VkkrjI8bB8fkza1XJCNv7PGKWILmQ/qUBvImsmHJejhK2CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8117fe3991705a8122abea40cd8f593b6ad118161a0c9fa6a0b77d165762e2e0","last_reissued_at":"2026-07-05T12:07:43.105332Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:43.105332Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TA-VLA: Elucidating the Design Space of Torque-aware Vision-Language-Action Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chenghao Yue, Haobo Xu, Hao Zhao, Huan-ang Gao, Zehao Lin, Zhuo Yang, Ziwei Wang, Zongzheng Zhang","submitted_at":"2025-09-09T17:50:37Z","abstract_excerpt":"Many robotic manipulation tasks require sensing and responding to force signals such as torque to assess whether the task has been successfully completed and to enable closed-loop control. However, current Vision-Language-Action (VLA) models lack the ability to integrate such subtle physical feedback. In this work, we explore Torque-aware VLA models, aiming to bridge this gap by systematically studying the design space for incorporating torque signals into existing VLA architectures. We identify and evaluate several strategies, leading to three key findings. First, introducing torque adapters "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.07962","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.07962/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.07962","created_at":"2026-07-05T12:07:43.105399+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.07962v1","created_at":"2026-07-05T12:07:43.105399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.07962","created_at":"2026-07-05T12:07:43.105399+00:00"},{"alias_kind":"pith_short_12","alias_value":"QEL74OMROBNI","created_at":"2026-07-05T12:07:43.105399+00:00"},{"alias_kind":"pith_short_16","alias_value":"QEL74OMROBNICIVL","created_at":"2026-07-05T12:07:43.105399+00:00"},{"alias_kind":"pith_short_8","alias_value":"QEL74OMR","created_at":"2026-07-05T12:07:43.105399+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07287","citing_title":"TouchWorld: A Predictive and Reactive Tactile Foundation Model for Dexterous Manipulation","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07287","citing_title":"TouchWorld: A Predictive and Reactive Tactile Foundation Model for Dexterous Manipulation","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23686","citing_title":"LIBERO-Safety: A Comprehensive Benchmark for Physical and Semantic Safety in Vision-Language-Action Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12105","citing_title":"DAM-VLA: Decoupled Asynchronous Multimodal Vision Language Action model","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09337","citing_title":"TORL-VLA: Tactile Guided Online Reinforcement Learning for Contact-Rich Manipulation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08737","citing_title":"Dream-Tac: A Unified Tactile World Action Model for Contact-Rich Robot Manipulation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23686","citing_title":"LIBERO-Safety: A Comprehensive Benchmark for Physical and Semantic Safety in Vision-Language-Action Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27251","citing_title":"Advancing Omnimodal Embodied Agents from Isolated Skills to Everyday Physical Autonomy","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07308","citing_title":"AT-VLA: Adaptive Tactile Injection for Enhanced Feedback Reaction in Vision-Language-Action Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18722","citing_title":"Dexora: Open-source VLA for High-DoF Bimanual Dexterity","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2512.09928","citing_title":"HiF-VLA: Hindsight, Insight and Foresight through Motion Representation for Vision-Language-Action Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":272,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03269","citing_title":"RLDX-1 Technical Report","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07308","citing_title":"AT-VLA: Adaptive Tactile Injection for Enhanced Feedback Reaction in Vision-Language-Action Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03269","citing_title":"RLDX-1 Technical Report","ref_index":119,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN","json":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN.json","graph_json":"https://pith.science/api/pith-number/QEL74OMROBNICIVL5JAM3D2ZHN/graph.json","events_json":"https://pith.science/api/pith-number/QEL74OMROBNICIVL5JAM3D2ZHN/events.json","paper":"https://pith.science/paper/QEL74OMR"},"agent_actions":{"view_html":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN","download_json":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN.json","view_paper":"https://pith.science/paper/QEL74OMR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.07962&json=true","fetch_graph":"https://pith.science/api/pith-number/QEL74OMROBNICIVL5JAM3D2ZHN/graph.json","fetch_events":"https://pith.science/api/pith-number/QEL74OMROBNICIVL5JAM3D2ZHN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN/action/storage_attestation","attest_author":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN/action/author_attestation","sign_citation":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN/action/citation_signature","submit_replication":"https://pith.science/pith/QEL74OMROBNICIVL5JAM3D2ZHN/action/replication_record"}},"created_at":"2026-07-05T12:07:43.105399+00:00","updated_at":"2026-07-05T12:07:43.105399+00:00"}