{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6FJ4VYGOFELCFNBNQEXENSDNIF","short_pith_number":"pith:6FJ4VYGO","schema_version":"1.0","canonical_sha256":"f153cae0ce291622b42d812e46c86d41452ff88dcdb9aea560cca9503a0fd95c","source":{"kind":"arxiv","id":"2104.03502","version":1},"attestation_state":"computed","paper":{"title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Leonardo Pepino, Luciana Ferrer, Pablo Riera","submitted_at":"2021-04-08T04:31:58Z","abstract_excerpt":"Emotion recognition datasets are relatively small, making the use of the more sophisticated deep learning approaches challenging. In this work, we propose a transfer learning method for speech emotion recognition where features extracted from pre-trained wav2vec 2.0 models are modeled using simple neural networks. We propose to combine the output of several layers from the pre-trained model using trainable weights which are learned jointly with the downstream model. Further, we compare performance using two different wav2vec 2.0 models, with and without finetuning for speech recognition. We ev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.03502","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2021-04-08T04:31:58Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"30f12a99abcf7b856b3107a34a9dcdbbc9e78fd9a00bcf9287cbd41ac745889c","abstract_canon_sha256":"03fffeb13f27a784555eb31fae6b65cc03e5ab05b5aa8676e678c9e50281cd95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:30:18.155510Z","signature_b64":"ik9WDyDKbMw0cbEAva9z28HuYy0UwHcF6mTiJ2Tfq720V7gaXa8lDFlFkwg48m6QQ1X/7c3a6gEyh845gpAQBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f153cae0ce291622b42d812e46c86d41452ff88dcdb9aea560cca9503a0fd95c","last_reissued_at":"2026-07-05T02:30:18.155109Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:30:18.155109Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Leonardo Pepino, Luciana Ferrer, Pablo Riera","submitted_at":"2021-04-08T04:31:58Z","abstract_excerpt":"Emotion recognition datasets are relatively small, making the use of the more sophisticated deep learning approaches challenging. In this work, we propose a transfer learning method for speech emotion recognition where features extracted from pre-trained wav2vec 2.0 models are modeled using simple neural networks. We propose to combine the output of several layers from the pre-trained model using trainable weights which are learned jointly with the downstream model. Further, we compare performance using two different wav2vec 2.0 models, with and without finetuning for speech recognition. We ev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.03502","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.03502/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.03502","created_at":"2026-07-05T02:30:18.155162+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.03502v1","created_at":"2026-07-05T02:30:18.155162+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.03502","created_at":"2026-07-05T02:30:18.155162+00:00"},{"alias_kind":"pith_short_12","alias_value":"6FJ4VYGOFELC","created_at":"2026-07-05T02:30:18.155162+00:00"},{"alias_kind":"pith_short_16","alias_value":"6FJ4VYGOFELCFNBN","created_at":"2026-07-05T02:30:18.155162+00:00"},{"alias_kind":"pith_short_8","alias_value":"6FJ4VYGO","created_at":"2026-07-05T02:30:18.155162+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06392","citing_title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22177","citing_title":"How Well Do Self-Supervised Speech Models Encode Age and Gender in Children's Speech? A Layer-Wise Analysis Across Multiple Architectures","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30550","citing_title":"SIGMA: Saliency-Guided Sparse Mask Attacks for Speech Emotion Recognition","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21869","citing_title":"Two-Stage Multimodal Framework for Emotion Mimicry Intensity Prediction","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19630","citing_title":"EMO-BOOST: Emotion-Augmented Audio-Visual Features for Improved Generalization in Deepfake Detection","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF","json":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF.json","graph_json":"https://pith.science/api/pith-number/6FJ4VYGOFELCFNBNQEXENSDNIF/graph.json","events_json":"https://pith.science/api/pith-number/6FJ4VYGOFELCFNBNQEXENSDNIF/events.json","paper":"https://pith.science/paper/6FJ4VYGO"},"agent_actions":{"view_html":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF","download_json":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF.json","view_paper":"https://pith.science/paper/6FJ4VYGO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.03502&json=true","fetch_graph":"https://pith.science/api/pith-number/6FJ4VYGOFELCFNBNQEXENSDNIF/graph.json","fetch_events":"https://pith.science/api/pith-number/6FJ4VYGOFELCFNBNQEXENSDNIF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF/action/storage_attestation","attest_author":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF/action/author_attestation","sign_citation":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF/action/citation_signature","submit_replication":"https://pith.science/pith/6FJ4VYGOFELCFNBNQEXENSDNIF/action/replication_record"}},"created_at":"2026-07-05T02:30:18.155162+00:00","updated_at":"2026-07-05T02:30:18.155162+00:00"}