{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:C7ZPQRVMT4ETA72CTSM2RD3WPN","short_pith_number":"pith:C7ZPQRVM","schema_version":"1.0","canonical_sha256":"17f2f846ac9f09307f429c99a88f767b5576e2f20667399e9fd759c4b15faafe","source":{"kind":"arxiv","id":"2505.22566","version":1},"attestation_state":"computed","paper":{"title":"Universal Visuo-Tactile Video Understanding for Embodied Interaction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fei Ma, Fei Richard Yu, Guangyu Chen, Mingyang Li, Shoujie Li, Wenbo Ding, Xingting Li, Yifan Xie","submitted_at":"2025-05-28T16:43:01Z","abstract_excerpt":"Tactile perception is essential for embodied agents to understand physical attributes of objects that cannot be determined through visual inspection alone. While existing approaches have made progress in visual and language modalities for physical understanding, they fail to effectively incorporate tactile information that provides crucial haptic feedback for real-world interaction. In this paper, we present VTV-LLM, the first multi-modal large language model for universal Visuo-Tactile Video (VTV) understanding that bridges the gap between tactile perception and natural language. To address t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22566","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-28T16:43:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a2f3eb9190395afa59087b25fbdd7b34db280ab086845f6591648abc67984020","abstract_canon_sha256":"3a9c3708c8fb6a38be7a2596d8c11873d3194135b04c5b87ab52e8befd6b65eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:26.801906Z","signature_b64":"3/QDnYXx8RvNc8MpsKhDzrUzGUfzXSqTtbxMFMm0GrPgqMghqPzTfBWjXBC4mR/jhIC/ylhNhTyprkThaRpHAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"17f2f846ac9f09307f429c99a88f767b5576e2f20667399e9fd759c4b15faafe","last_reissued_at":"2026-07-05T11:11:26.801355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:26.801355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Universal Visuo-Tactile Video Understanding for Embodied Interaction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fei Ma, Fei Richard Yu, Guangyu Chen, Mingyang Li, Shoujie Li, Wenbo Ding, Xingting Li, Yifan Xie","submitted_at":"2025-05-28T16:43:01Z","abstract_excerpt":"Tactile perception is essential for embodied agents to understand physical attributes of objects that cannot be determined through visual inspection alone. While existing approaches have made progress in visual and language modalities for physical understanding, they fail to effectively incorporate tactile information that provides crucial haptic feedback for real-world interaction. In this paper, we present VTV-LLM, the first multi-modal large language model for universal Visuo-Tactile Video (VTV) understanding that bridges the gap between tactile perception and natural language. To address t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22566","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22566","created_at":"2026-07-05T11:11:26.801416+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22566v1","created_at":"2026-07-05T11:11:26.801416+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22566","created_at":"2026-07-05T11:11:26.801416+00:00"},{"alias_kind":"pith_short_12","alias_value":"C7ZPQRVMT4ET","created_at":"2026-07-05T11:11:26.801416+00:00"},{"alias_kind":"pith_short_16","alias_value":"C7ZPQRVMT4ETA72C","created_at":"2026-07-05T11:11:26.801416+00:00"},{"alias_kind":"pith_short_8","alias_value":"C7ZPQRVM","created_at":"2026-07-05T11:11:26.801416+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31451","citing_title":"UniTac: A Unified Multimodal Model for Cross-Sensor Tactile Understanding and Generation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24681","citing_title":"Learning Human-Intention Priors from Large-Scale Human Demonstrations for Robotic Manipulation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17336","citing_title":"Tactile-based Multimodal Fusion in Embodied Intelligence: A Survey of Vision, Language, and Contact-Driven Paradigms","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24681","citing_title":"Learning Human-Intention Priors from Large-Scale Human Demonstrations for Robotic Manipulation","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN","json":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN.json","graph_json":"https://pith.science/api/pith-number/C7ZPQRVMT4ETA72CTSM2RD3WPN/graph.json","events_json":"https://pith.science/api/pith-number/C7ZPQRVMT4ETA72CTSM2RD3WPN/events.json","paper":"https://pith.science/paper/C7ZPQRVM"},"agent_actions":{"view_html":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN","download_json":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN.json","view_paper":"https://pith.science/paper/C7ZPQRVM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22566&json=true","fetch_graph":"https://pith.science/api/pith-number/C7ZPQRVMT4ETA72CTSM2RD3WPN/graph.json","fetch_events":"https://pith.science/api/pith-number/C7ZPQRVMT4ETA72CTSM2RD3WPN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN/action/storage_attestation","attest_author":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN/action/author_attestation","sign_citation":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN/action/citation_signature","submit_replication":"https://pith.science/pith/C7ZPQRVMT4ETA72CTSM2RD3WPN/action/replication_record"}},"created_at":"2026-07-05T11:11:26.801416+00:00","updated_at":"2026-07-05T11:11:26.801416+00:00"}