{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7AIZDHTCULVH6KCAMOKF7VYHGX","short_pith_number":"pith:7AIZDHTC","schema_version":"1.0","canonical_sha256":"f811919e62a2ea7f284063945fd70735fb2fc51664b07d4ead3c054151875536","source":{"kind":"arxiv","id":"2507.22934","version":1},"attestation_state":"computed","paper":{"title":"Deep Learning Approaches for Multimodal Intent Recognition: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jianhua Tao, Jingwei Zhao, Jingyao Xue, Junyang Wu, Minchi Hu, Qifei Li, Ya Li, Yingming Gao, Yingying Zhou, Yuhua Wen, Zhengqi Wen","submitted_at":"2025-07-24T17:12:01Z","abstract_excerpt":"Intent recognition aims to identify users' underlying intentions, traditionally focusing on text in natural language processing. With growing demands for natural human-computer interaction, the field has evolved through deep learning and multimodal approaches, incorporating data from audio, vision, and physiological signals. Recently, the introduction of Transformer-based models has led to notable breakthroughs in this domain. This article surveys deep learning methods for intent recognition, covering the shift from unimodal to multimodal techniques, relevant datasets, methodologies, applicati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22934","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-24T17:12:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"14e749c08057bdcb070c13f2515f643e9453467e57e94f3e1d61a38640962b3e","abstract_canon_sha256":"dc45365e21b2e4465ef21da07480d8460c33b7706ff557191ebb40126133f229"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:02.260437Z","signature_b64":"Iq+Rl4p9w5wvjVLB5a/3uISsei7DKcwpsDX9fy6X8nCpSbCtxvUBaZLBbl7D8AI4qpH5fsamsVyPlotusIOWCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f811919e62a2ea7f284063945fd70735fb2fc51664b07d4ead3c054151875536","last_reissued_at":"2026-07-05T11:46:02.259912Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:02.259912Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Learning Approaches for Multimodal Intent Recognition: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jianhua Tao, Jingwei Zhao, Jingyao Xue, Junyang Wu, Minchi Hu, Qifei Li, Ya Li, Yingming Gao, Yingying Zhou, Yuhua Wen, Zhengqi Wen","submitted_at":"2025-07-24T17:12:01Z","abstract_excerpt":"Intent recognition aims to identify users' underlying intentions, traditionally focusing on text in natural language processing. With growing demands for natural human-computer interaction, the field has evolved through deep learning and multimodal approaches, incorporating data from audio, vision, and physiological signals. Recently, the introduction of Transformer-based models has led to notable breakthroughs in this domain. This article surveys deep learning methods for intent recognition, covering the shift from unimodal to multimodal techniques, relevant datasets, methodologies, applicati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22934","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22934","created_at":"2026-07-05T11:46:02.259970+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22934v1","created_at":"2026-07-05T11:46:02.259970+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22934","created_at":"2026-07-05T11:46:02.259970+00:00"},{"alias_kind":"pith_short_12","alias_value":"7AIZDHTCULVH","created_at":"2026-07-05T11:46:02.259970+00:00"},{"alias_kind":"pith_short_16","alias_value":"7AIZDHTCULVH6KCA","created_at":"2026-07-05T11:46:02.259970+00:00"},{"alias_kind":"pith_short_8","alias_value":"7AIZDHTC","created_at":"2026-07-05T11:46:02.259970+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24002","citing_title":"IntentVLM: Open-Vocabulary Intention Recognition through Forward-Inverse Modeling with Video-Language Models","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX","json":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX.json","graph_json":"https://pith.science/api/pith-number/7AIZDHTCULVH6KCAMOKF7VYHGX/graph.json","events_json":"https://pith.science/api/pith-number/7AIZDHTCULVH6KCAMOKF7VYHGX/events.json","paper":"https://pith.science/paper/7AIZDHTC"},"agent_actions":{"view_html":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX","download_json":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX.json","view_paper":"https://pith.science/paper/7AIZDHTC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22934&json=true","fetch_graph":"https://pith.science/api/pith-number/7AIZDHTCULVH6KCAMOKF7VYHGX/graph.json","fetch_events":"https://pith.science/api/pith-number/7AIZDHTCULVH6KCAMOKF7VYHGX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX/action/storage_attestation","attest_author":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX/action/author_attestation","sign_citation":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX/action/citation_signature","submit_replication":"https://pith.science/pith/7AIZDHTCULVH6KCAMOKF7VYHGX/action/replication_record"}},"created_at":"2026-07-05T11:46:02.259970+00:00","updated_at":"2026-07-05T11:46:02.259970+00:00"}