{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HUNPC2VPO3NNAJQJKLD4X6OYCA","short_pith_number":"pith:HUNPC2VP","schema_version":"1.0","canonical_sha256":"3d1af16aaf76dad0260952c7cbf9d8102d1345bc3f33ff58433ac13243780f02","source":{"kind":"arxiv","id":"2312.15185","version":1},"attestation_state":"computed","paper":{"title":"emotion2vec: Self-Supervised Pre-Training for Speech Emotion Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Jiaxin Ye, Jinchao Li, Shiliang Zhang, Xie Chen, Zhifu Gao, Zhisheng Zheng, Ziyang Ma","submitted_at":"2023-12-23T07:46:55Z","abstract_excerpt":"We propose emotion2vec, a universal speech emotion representation model. emotion2vec is pre-trained on open-source unlabeled emotion data through self-supervised online distillation, combining utterance-level loss and frame-level loss during pre-training. emotion2vec outperforms state-of-the-art pre-trained universal models and emotion specialist models by only training linear layers for the speech emotion recognition task on the mainstream IEMOCAP dataset. In addition, emotion2vec shows consistent improvements among 10 different languages of speech emotion recognition datasets. emotion2vec al"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15185","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-23T07:46:55Z","cross_cats_sorted":["cs.HC","cs.MM","cs.SD","eess.AS"],"title_canon_sha256":"597f6ddf19377a26fc436323a2018ce449809796ff0cbd4c708b0a5e8687d7dc","abstract_canon_sha256":"ccc6a97862c314a53cc5690a57ccbb6911274b1132579e7e98d381f3b7a5ab23"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:27:32.024932Z","signature_b64":"o8aAzpcIzuKvz5pRD2Yc1aG8H8co+7bXVrCfBIQBLss67PdS3GSaDWFzjczjC4l/UsBLhZdaQO+Y58Pw1KY5Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d1af16aaf76dad0260952c7cbf9d8102d1345bc3f33ff58433ac13243780f02","last_reissued_at":"2026-07-05T07:27:32.024462Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:27:32.024462Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"emotion2vec: Self-Supervised Pre-Training for Speech Emotion Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.HC","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Jiaxin Ye, Jinchao Li, Shiliang Zhang, Xie Chen, Zhifu Gao, Zhisheng Zheng, Ziyang Ma","submitted_at":"2023-12-23T07:46:55Z","abstract_excerpt":"We propose emotion2vec, a universal speech emotion representation model. emotion2vec is pre-trained on open-source unlabeled emotion data through self-supervised online distillation, combining utterance-level loss and frame-level loss during pre-training. emotion2vec outperforms state-of-the-art pre-trained universal models and emotion specialist models by only training linear layers for the speech emotion recognition task on the mainstream IEMOCAP dataset. In addition, emotion2vec shows consistent improvements among 10 different languages of speech emotion recognition datasets. emotion2vec al"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15185","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15185/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15185","created_at":"2026-07-05T07:27:32.024521+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15185v1","created_at":"2026-07-05T07:27:32.024521+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15185","created_at":"2026-07-05T07:27:32.024521+00:00"},{"alias_kind":"pith_short_12","alias_value":"HUNPC2VPO3NN","created_at":"2026-07-05T07:27:32.024521+00:00"},{"alias_kind":"pith_short_16","alias_value":"HUNPC2VPO3NNAJQJ","created_at":"2026-07-05T07:27:32.024521+00:00"},{"alias_kind":"pith_short_8","alias_value":"HUNPC2VP","created_at":"2026-07-05T07:27:32.024521+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17339","citing_title":"SpeechDx: A Multi-Task Benchmark for Clinical Speech AI","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31247","citing_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","ref_index":189,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31128","citing_title":"UniSAE: Unified Speech Attribute Editing on Speaker, Emotion and Low-Level Content via Discrete Phonetic Posteriorgram Modelling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18512","citing_title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22732","citing_title":"Beyond Acoustic Emotion Recognition: Multimodal Pathos Analysis in Political Speech Using LLM-Based and Acoustic Emotion Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13293","citing_title":"Cross-modal Consistency Guidance for Robust Emotion Control in Auto-Regressive TTS Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19630","citing_title":"EMO-BOOST: Emotion-Augmented Audio-Visual Features for Improved Generalization in Deepfake Detection","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23511","citing_title":"MECAT: A Multi-Experts Constructed Benchmark for Fine-Grained Audio Understanding Tasks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13293","citing_title":"Cross-modal Consistency Guidance for Robust Emotion Control in Auto-Regressive TTS Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26417","citing_title":"EmoTransCap: Dataset and Pipeline for Emotion Transition-Aware Speech Captioning in Discourses","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08363","citing_title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06765","citing_title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA","json":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA.json","graph_json":"https://pith.science/api/pith-number/HUNPC2VPO3NNAJQJKLD4X6OYCA/graph.json","events_json":"https://pith.science/api/pith-number/HUNPC2VPO3NNAJQJKLD4X6OYCA/events.json","paper":"https://pith.science/paper/HUNPC2VP"},"agent_actions":{"view_html":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA","download_json":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA.json","view_paper":"https://pith.science/paper/HUNPC2VP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15185&json=true","fetch_graph":"https://pith.science/api/pith-number/HUNPC2VPO3NNAJQJKLD4X6OYCA/graph.json","fetch_events":"https://pith.science/api/pith-number/HUNPC2VPO3NNAJQJKLD4X6OYCA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA/action/storage_attestation","attest_author":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA/action/author_attestation","sign_citation":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA/action/citation_signature","submit_replication":"https://pith.science/pith/HUNPC2VPO3NNAJQJKLD4X6OYCA/action/replication_record"}},"created_at":"2026-07-05T07:27:32.024521+00:00","updated_at":"2026-07-05T07:27:32.024521+00:00"}