{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XUPGNB5VFJQNCCFE7D2OGTQOON","short_pith_number":"pith:XUPGNB5V","schema_version":"1.0","canonical_sha256":"bd1e6687b52a60d108a4f8f4e34e0e737547bf61e40a9be2151732481f3514a8","source":{"kind":"arxiv","id":"2403.09813","version":3},"attestation_state":"computed","paper":{"title":"Towards Comprehensive Multimodal Perception: Introducing the Touch-Language-Vision Dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Bin Fang, Jinan Xu, Jing Gao, Ning Cheng, Wenjuan Han, You Li","submitted_at":"2024-03-14T19:01:54Z","abstract_excerpt":"Tactility provides crucial support and enhancement for the perception and interaction capabilities of both humans and robots. Nevertheless, the multimodal research related to touch primarily focuses on visual and tactile modalities, with limited exploration in the domain of language. Beyond vocabulary, sentence-level descriptions contain richer semantics. Based on this, we construct a touch-language-vision dataset named TLV (Touch-Language-Vision) by human-machine cascade collaboration, featuring sentence-level descriptions for multimode alignment. The new dataset is used to fine-tune our prop"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.09813","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-14T19:01:54Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"b1ec9a90fa385edcd7d394ef7a38f31727db0c55e07e4935330cd0aa63f9fb92","abstract_canon_sha256":"2188fddf89d2ad130d8c38712d1c2f4ddd58df05edf284d849c6ad3d43beab28"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:38.560906Z","signature_b64":"BoIZoG47tOa2OZNxmqNo5V3yn+pjq8ycG0q0YHbhM9j/BdXDxWeMZBXlC0STK5F2A4gZApdnRQNmNTx9DFBPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bd1e6687b52a60d108a4f8f4e34e0e737547bf61e40a9be2151732481f3514a8","last_reissued_at":"2026-07-05T08:32:38.560402Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:38.560402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Comprehensive Multimodal Perception: Introducing the Touch-Language-Vision Dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Bin Fang, Jinan Xu, Jing Gao, Ning Cheng, Wenjuan Han, You Li","submitted_at":"2024-03-14T19:01:54Z","abstract_excerpt":"Tactility provides crucial support and enhancement for the perception and interaction capabilities of both humans and robots. Nevertheless, the multimodal research related to touch primarily focuses on visual and tactile modalities, with limited exploration in the domain of language. Beyond vocabulary, sentence-level descriptions contain richer semantics. Based on this, we construct a touch-language-vision dataset named TLV (Touch-Language-Vision) by human-machine cascade collaboration, featuring sentence-level descriptions for multimode alignment. The new dataset is used to fine-tune our prop"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.09813","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.09813/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.09813","created_at":"2026-07-05T08:32:38.560466+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.09813v3","created_at":"2026-07-05T08:32:38.560466+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.09813","created_at":"2026-07-05T08:32:38.560466+00:00"},{"alias_kind":"pith_short_12","alias_value":"XUPGNB5VFJQN","created_at":"2026-07-05T08:32:38.560466+00:00"},{"alias_kind":"pith_short_16","alias_value":"XUPGNB5VFJQNCCFE","created_at":"2026-07-05T08:32:38.560466+00:00"},{"alias_kind":"pith_short_8","alias_value":"XUPGNB5V","created_at":"2026-07-05T08:32:38.560466+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31694","citing_title":"RCT: A Robot-Collected Touch-Vision-Language Dataset for Tactile Generalization","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17336","citing_title":"Tactile-based Multimodal Fusion in Embodied Intelligence: A Survey of Vision, Language, and Contact-Driven Paradigms","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON","json":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON.json","graph_json":"https://pith.science/api/pith-number/XUPGNB5VFJQNCCFE7D2OGTQOON/graph.json","events_json":"https://pith.science/api/pith-number/XUPGNB5VFJQNCCFE7D2OGTQOON/events.json","paper":"https://pith.science/paper/XUPGNB5V"},"agent_actions":{"view_html":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON","download_json":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON.json","view_paper":"https://pith.science/paper/XUPGNB5V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.09813&json=true","fetch_graph":"https://pith.science/api/pith-number/XUPGNB5VFJQNCCFE7D2OGTQOON/graph.json","fetch_events":"https://pith.science/api/pith-number/XUPGNB5VFJQNCCFE7D2OGTQOON/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON/action/storage_attestation","attest_author":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON/action/author_attestation","sign_citation":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON/action/citation_signature","submit_replication":"https://pith.science/pith/XUPGNB5VFJQNCCFE7D2OGTQOON/action/replication_record"}},"created_at":"2026-07-05T08:32:38.560466+00:00","updated_at":"2026-07-05T08:32:38.560466+00:00"}