{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3ZGLXLKZWKG5GX3FPJ7YXBHTLX","short_pith_number":"pith:3ZGLXLKZ","schema_version":"1.0","canonical_sha256":"de4cbbad59b28dd35f657a7f8b84f35de2cd001164540122aa830b84cf12c274","source":{"kind":"arxiv","id":"2308.07145","version":1},"attestation_state":"computed","paper":{"title":"Integrating Emotion Recognition with Speech Recognition and Speaker Diarisation for Conversations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Chao Zhang, Philip C. Woodland, Wen Wu","submitted_at":"2023-08-14T13:50:47Z","abstract_excerpt":"Although automatic emotion recognition (AER) has recently drawn significant research interest, most current AER studies use manually segmented utterances, which are usually unavailable for dialogue systems. This paper proposes integrating AER with automatic speech recognition (ASR) and speaker diarisation (SD) in a jointly-trained system. Distinct output layers are built for four sub-tasks including AER, ASR, voice activity detection and speaker classification based on a shared encoder. Taking the audio of a conversation as input, the integrated system finds all speech segments and transcribes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.07145","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"eess.AS","submitted_at":"2023-08-14T13:50:47Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"acc4de109e227462be3701219297439a6d5198729302640e47471af033f8d21c","abstract_canon_sha256":"e5c543866e091c3a78134a5a94d02901860fbf446ad6de37846f31717282c55b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:40:56.719742Z","signature_b64":"5sPAjWro1jWpdaT5WreBnXjXhN8+YrzdbJizpe7dLxC0LOUsz2Ucou3XWrGrjYH3bwhFa7rrkEPVesEui/IxCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de4cbbad59b28dd35f657a7f8b84f35de2cd001164540122aa830b84cf12c274","last_reissued_at":"2026-07-05T06:40:56.719276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:40:56.719276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Integrating Emotion Recognition with Speech Recognition and Speaker Diarisation for Conversations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Chao Zhang, Philip C. Woodland, Wen Wu","submitted_at":"2023-08-14T13:50:47Z","abstract_excerpt":"Although automatic emotion recognition (AER) has recently drawn significant research interest, most current AER studies use manually segmented utterances, which are usually unavailable for dialogue systems. This paper proposes integrating AER with automatic speech recognition (ASR) and speaker diarisation (SD) in a jointly-trained system. Distinct output layers are built for four sub-tasks including AER, ASR, voice activity detection and speaker classification based on a shared encoder. Taking the audio of a conversation as input, the integrated system finds all speech segments and transcribes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.07145","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.07145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.07145","created_at":"2026-07-05T06:40:56.719343+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.07145v1","created_at":"2026-07-05T06:40:56.719343+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.07145","created_at":"2026-07-05T06:40:56.719343+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ZGLXLKZWKG5","created_at":"2026-07-05T06:40:56.719343+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ZGLXLKZWKG5GX3F","created_at":"2026-07-05T06:40:56.719343+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ZGLXLKZ","created_at":"2026-07-05T06:40:56.719343+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.25776","citing_title":"Unrequited Emotions: Investigating the Gaps in Motivation and Practice in Speech Emotion Recognition Research","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX","json":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX.json","graph_json":"https://pith.science/api/pith-number/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/graph.json","events_json":"https://pith.science/api/pith-number/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/events.json","paper":"https://pith.science/paper/3ZGLXLKZ"},"agent_actions":{"view_html":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX","download_json":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX.json","view_paper":"https://pith.science/paper/3ZGLXLKZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.07145&json=true","fetch_graph":"https://pith.science/api/pith-number/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/graph.json","fetch_events":"https://pith.science/api/pith-number/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/action/storage_attestation","attest_author":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/action/author_attestation","sign_citation":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/action/citation_signature","submit_replication":"https://pith.science/pith/3ZGLXLKZWKG5GX3FPJ7YXBHTLX/action/replication_record"}},"created_at":"2026-07-05T06:40:56.719343+00:00","updated_at":"2026-07-05T06:40:56.719343+00:00"}