{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DSBRVIC2NNFDCJYTRHWF2EYD7U","short_pith_number":"pith:DSBRVIC2","schema_version":"1.0","canonical_sha256":"1c831aa05a6b4a31271389ec5d1303fd1d1aa5c03735ade8bfd2271ee933c371","source":{"kind":"arxiv","id":"2203.07378","version":4},"attestation_state":"computed","paper":{"title":"Dawn of the transformer era in speech emotion recognition: closing the valence gap","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Andreas Triantafyllopoulos, Bj\\\"orn W. Schuller, Felix Burkhardt, Florian Eyben, Hagen Wierstorf, Johannes Wagner, Maximilian Schmitt","submitted_at":"2022-03-14T13:21:47Z","abstract_excerpt":"Recent advances in transformer-based architectures which are pre-trained in self-supervised manner have shown great promise in several machine learning tasks. In the audio domain, such architectures have also been successfully utilised in the field of speech emotion recognition (SER). However, existing works have not evaluated the influence of model size and pre-training data on downstream performance, and have shown limited attention to generalisation, robustness, fairness, and efficiency. The present contribution conducts a thorough analysis of these aspects on several pre-trained variants o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.07378","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"eess.AS","submitted_at":"2022-03-14T13:21:47Z","cross_cats_sorted":["cs.LG","cs.SD"],"title_canon_sha256":"fc8f0ccea9c93804a85bbc6152d27f631631a57118aef7019acf9b51fcaf04ae","abstract_canon_sha256":"71dc01166adaf2c12981b9b7fd3ad8ce049bf9e2f9e1d9aea69d5a4228dc64e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:44.607252Z","signature_b64":"2Y19J4/Fq+DtZ36IYXUzH+nA9l9HQgvc0A42cmf2fUA0u4rLME51J2XZ5mrTf9yJZqjopNh3KtQfTAqr+NIDBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c831aa05a6b4a31271389ec5d1303fd1d1aa5c03735ade8bfd2271ee933c371","last_reissued_at":"2026-07-05T06:48:44.606725Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:44.606725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dawn of the transformer era in speech emotion recognition: closing the valence gap","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Andreas Triantafyllopoulos, Bj\\\"orn W. Schuller, Felix Burkhardt, Florian Eyben, Hagen Wierstorf, Johannes Wagner, Maximilian Schmitt","submitted_at":"2022-03-14T13:21:47Z","abstract_excerpt":"Recent advances in transformer-based architectures which are pre-trained in self-supervised manner have shown great promise in several machine learning tasks. In the audio domain, such architectures have also been successfully utilised in the field of speech emotion recognition (SER). However, existing works have not evaluated the influence of model size and pre-training data on downstream performance, and have shown limited attention to generalisation, robustness, fairness, and efficiency. The present contribution conducts a thorough analysis of these aspects on several pre-trained variants o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.07378","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.07378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.07378","created_at":"2026-07-05T06:48:44.606798+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.07378v4","created_at":"2026-07-05T06:48:44.606798+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.07378","created_at":"2026-07-05T06:48:44.606798+00:00"},{"alias_kind":"pith_short_12","alias_value":"DSBRVIC2NNFD","created_at":"2026-07-05T06:48:44.606798+00:00"},{"alias_kind":"pith_short_16","alias_value":"DSBRVIC2NNFDCJYT","created_at":"2026-07-05T06:48:44.606798+00:00"},{"alias_kind":"pith_short_8","alias_value":"DSBRVIC2","created_at":"2026-07-05T06:48:44.606798+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03283","citing_title":"SpeakerCard-1M: An Evidence-Grounded Corpus for In-the-Wild Speaker Verification","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21869","citing_title":"Two-Stage Multimodal Framework for Emotion Mimicry Intensity Prediction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12387","citing_title":"A Semi-Supervised Framework for Speech Confidence Detection using Whisper","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U","json":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U.json","graph_json":"https://pith.science/api/pith-number/DSBRVIC2NNFDCJYTRHWF2EYD7U/graph.json","events_json":"https://pith.science/api/pith-number/DSBRVIC2NNFDCJYTRHWF2EYD7U/events.json","paper":"https://pith.science/paper/DSBRVIC2"},"agent_actions":{"view_html":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U","download_json":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U.json","view_paper":"https://pith.science/paper/DSBRVIC2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.07378&json=true","fetch_graph":"https://pith.science/api/pith-number/DSBRVIC2NNFDCJYTRHWF2EYD7U/graph.json","fetch_events":"https://pith.science/api/pith-number/DSBRVIC2NNFDCJYTRHWF2EYD7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U/action/storage_attestation","attest_author":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U/action/author_attestation","sign_citation":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U/action/citation_signature","submit_replication":"https://pith.science/pith/DSBRVIC2NNFDCJYTRHWF2EYD7U/action/replication_record"}},"created_at":"2026-07-05T06:48:44.606798+00:00","updated_at":"2026-07-05T06:48:44.606798+00:00"}