{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:C4PVJL6CRIWB6ZQVUWB5LJCXT6","short_pith_number":"pith:C4PVJL6C","schema_version":"1.0","canonical_sha256":"171f54afc28a2c1f6615a583d5a4579fbb76e163fe69b8eb1036e32c58031f8b","source":{"kind":"arxiv","id":"2207.12895","version":1},"attestation_state":"computed","paper":{"title":"Multimodal Speech Emotion Recognition using Cross Attention with Aligned Audio and Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Kyomin Jung, Seunghyun Yoon, Yoonhyung Lee","submitted_at":"2022-07-26T13:44:07Z","abstract_excerpt":"In this paper, we propose a novel speech emotion recognition model called Cross Attention Network (CAN) that uses aligned audio and text signals as inputs. It is inspired by the fact that humans recognize speech as a combination of simultaneously produced acoustic and textual signals. First, our method segments the audio and the underlying text signals into equal number of steps in an aligned way so that the same time steps of the sequential signals cover the same time span in the signals. Together with this technique, we apply the cross attention to aggregate the sequential information from t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.12895","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2022-07-26T13:44:07Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"c684062c1953182fdf4f06a03d0a7fdaeb0eb928760d9beda3a2fbc2b24f6f80","abstract_canon_sha256":"03093fc15a72c802e2f33a62ebd77accfd8e43c8c89e2d1e7d67791cf168a402"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:43:51.202333Z","signature_b64":"XVe6sGDSKh5Yp5lF6G+XJlqpPoypDr9ObS6EU+aDZoAzbMzLD39j+gPTqakkT60GBXwETUNo6lAYIkaLTU2VDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"171f54afc28a2c1f6615a583d5a4579fbb76e163fe69b8eb1036e32c58031f8b","last_reissued_at":"2026-07-05T04:43:51.201941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:43:51.201941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Speech Emotion Recognition using Cross Attention with Aligned Audio and Text","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Kyomin Jung, Seunghyun Yoon, Yoonhyung Lee","submitted_at":"2022-07-26T13:44:07Z","abstract_excerpt":"In this paper, we propose a novel speech emotion recognition model called Cross Attention Network (CAN) that uses aligned audio and text signals as inputs. It is inspired by the fact that humans recognize speech as a combination of simultaneously produced acoustic and textual signals. First, our method segments the audio and the underlying text signals into equal number of steps in an aligned way so that the same time steps of the sequential signals cover the same time span in the signals. Together with this technique, we apply the cross attention to aggregate the sequential information from t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.12895","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.12895/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.12895","created_at":"2026-07-05T04:43:51.201996+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.12895v1","created_at":"2026-07-05T04:43:51.201996+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.12895","created_at":"2026-07-05T04:43:51.201996+00:00"},{"alias_kind":"pith_short_12","alias_value":"C4PVJL6CRIWB","created_at":"2026-07-05T04:43:51.201996+00:00"},{"alias_kind":"pith_short_16","alias_value":"C4PVJL6CRIWB6ZQV","created_at":"2026-07-05T04:43:51.201996+00:00"},{"alias_kind":"pith_short_8","alias_value":"C4PVJL6C","created_at":"2026-07-05T04:43:51.201996+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.07388","citing_title":"Recent Advances in Multimodal Affective Computing: An NLP Perspective","ref_index":204,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6","json":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6.json","graph_json":"https://pith.science/api/pith-number/C4PVJL6CRIWB6ZQVUWB5LJCXT6/graph.json","events_json":"https://pith.science/api/pith-number/C4PVJL6CRIWB6ZQVUWB5LJCXT6/events.json","paper":"https://pith.science/paper/C4PVJL6C"},"agent_actions":{"view_html":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6","download_json":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6.json","view_paper":"https://pith.science/paper/C4PVJL6C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.12895&json=true","fetch_graph":"https://pith.science/api/pith-number/C4PVJL6CRIWB6ZQVUWB5LJCXT6/graph.json","fetch_events":"https://pith.science/api/pith-number/C4PVJL6CRIWB6ZQVUWB5LJCXT6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6/action/storage_attestation","attest_author":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6/action/author_attestation","sign_citation":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6/action/citation_signature","submit_replication":"https://pith.science/pith/C4PVJL6CRIWB6ZQVUWB5LJCXT6/action/replication_record"}},"created_at":"2026-07-05T04:43:51.201996+00:00","updated_at":"2026-07-05T04:43:51.201996+00:00"}