{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DCCAVS6A7HVRIFT77D5QVWMIMR","short_pith_number":"pith:DCCAVS6A","schema_version":"1.0","canonical_sha256":"18840acbc0f9eb14167ff8fb0ad9886477862586f6a5cb096c3ed071dea47ffd","source":{"kind":"arxiv","id":"2507.19356","version":1},"attestation_state":"computed","paper":{"title":"Enhancing Speech Emotion Recognition Leveraging Aligning Timestamps of ASR Transcripts and Speaker Diarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Berlin Chen, Hsuan-Yu Wang, Pei-Ying Lee","submitted_at":"2025-07-25T15:05:20Z","abstract_excerpt":"In this paper, we investigate the impact of incorporating timestamp-based alignment between Automatic Speech Recognition (ASR) transcripts and Speaker Diarization (SD) outputs on Speech Emotion Recognition (SER) accuracy. Misalignment between these two modalities often reduces the reliability of multimodal emotion recognition systems, particularly in conversational contexts. To address this issue, we introduce an alignment pipeline utilizing pre-trained ASR and speaker diarization models, systematically synchronizing timestamps to generate accurately labeled speaker segments. Our multimodal ap"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.19356","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-25T15:05:20Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"d9a7925a66d4782a5e7aca124562ff8cf98d51772fc1b006e1f5c6d2b8d8bedb","abstract_canon_sha256":"80da816eeaecf7b0f837b5e9eb3dee6e30ebe3a2072f2c4eb16c61906398932f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:24.042086Z","signature_b64":"7BQZXtcyjTG6WAZ8bvaZe2RRF1m8+NYtHGXyWqrQLdeEwuqEh7lKLVeioJv7dVqgYadtjJxBu+KRFoiXvs0TCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18840acbc0f9eb14167ff8fb0ad9886477862586f6a5cb096c3ed071dea47ffd","last_reissued_at":"2026-07-05T11:43:24.041632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:24.041632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Speech Emotion Recognition Leveraging Aligning Timestamps of ASR Transcripts and Speaker Diarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Berlin Chen, Hsuan-Yu Wang, Pei-Ying Lee","submitted_at":"2025-07-25T15:05:20Z","abstract_excerpt":"In this paper, we investigate the impact of incorporating timestamp-based alignment between Automatic Speech Recognition (ASR) transcripts and Speaker Diarization (SD) outputs on Speech Emotion Recognition (SER) accuracy. Misalignment between these two modalities often reduces the reliability of multimodal emotion recognition systems, particularly in conversational contexts. To address this issue, we introduce an alignment pipeline utilizing pre-trained ASR and speaker diarization models, systematically synchronizing timestamps to generate accurately labeled speaker segments. Our multimodal ap"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.19356","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.19356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.19356","created_at":"2026-07-05T11:43:24.041689+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.19356v1","created_at":"2026-07-05T11:43:24.041689+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.19356","created_at":"2026-07-05T11:43:24.041689+00:00"},{"alias_kind":"pith_short_12","alias_value":"DCCAVS6A7HVR","created_at":"2026-07-05T11:43:24.041689+00:00"},{"alias_kind":"pith_short_16","alias_value":"DCCAVS6A7HVRIFT7","created_at":"2026-07-05T11:43:24.041689+00:00"},{"alias_kind":"pith_short_8","alias_value":"DCCAVS6A","created_at":"2026-07-05T11:43:24.041689+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR","json":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR.json","graph_json":"https://pith.science/api/pith-number/DCCAVS6A7HVRIFT77D5QVWMIMR/graph.json","events_json":"https://pith.science/api/pith-number/DCCAVS6A7HVRIFT77D5QVWMIMR/events.json","paper":"https://pith.science/paper/DCCAVS6A"},"agent_actions":{"view_html":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR","download_json":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR.json","view_paper":"https://pith.science/paper/DCCAVS6A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.19356&json=true","fetch_graph":"https://pith.science/api/pith-number/DCCAVS6A7HVRIFT77D5QVWMIMR/graph.json","fetch_events":"https://pith.science/api/pith-number/DCCAVS6A7HVRIFT77D5QVWMIMR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR/action/storage_attestation","attest_author":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR/action/author_attestation","sign_citation":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR/action/citation_signature","submit_replication":"https://pith.science/pith/DCCAVS6A7HVRIFT77D5QVWMIMR/action/replication_record"}},"created_at":"2026-07-05T11:43:24.041689+00:00","updated_at":"2026-07-05T11:43:24.041689+00:00"}