{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4BYJSDQEQA24YCL7YQA3THMDAL","short_pith_number":"pith:4BYJSDQE","schema_version":"1.0","canonical_sha256":"e070990e048035cc097fc401b99d8302cd0ccb44ddb39c52a1c56b7b79559991","source":{"kind":"arxiv","id":"2409.09511","version":1},"attestation_state":"computed","paper":{"title":"Explaining Deep Learning Embeddings for Speech Emotion Recognition by Predicting Interpretable Acoustic Features","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Daniel M. Low, Fabio Catania, Gasser Elbanna, Satrajit S. Ghosh, Satvik Dixit","submitted_at":"2024-09-14T19:18:56Z","abstract_excerpt":"Pre-trained deep learning embeddings have consistently shown superior performance over handcrafted acoustic features in speech emotion recognition (SER). However, unlike acoustic features with clear physical meaning, these embeddings lack clear interpretability. Explaining these embeddings is crucial for building trust in healthcare and security applications and advancing the scientific understanding of the acoustic information that is encoded in them. This paper proposes a modified probing approach to explain deep learning embeddings in the SER space. We predict interpretable acoustic feature"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.09511","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-09-14T19:18:56Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"88cde54206a20dc8218c4e1eeb10a4e85af9a655b960a5bbf9b5a20bd88f3bd8","abstract_canon_sha256":"e5f5d791c3c033582f3eb00a249dd1299103ee14ef42afbe4ee8a9e47495cac6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:10.690456Z","signature_b64":"CuQm4SyIicVFFQQGJgO6SUZwG12tOH8uVUq5X3ppe4ZFPewyzsT/CdWCaxbC7zZG6WeCh6aXFFoHyXiaXVq7BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e070990e048035cc097fc401b99d8302cd0ccb44ddb39c52a1c56b7b79559991","last_reissued_at":"2026-07-05T09:07:10.689858Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:10.689858Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Explaining Deep Learning Embeddings for Speech Emotion Recognition by Predicting Interpretable Acoustic Features","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Daniel M. Low, Fabio Catania, Gasser Elbanna, Satrajit S. Ghosh, Satvik Dixit","submitted_at":"2024-09-14T19:18:56Z","abstract_excerpt":"Pre-trained deep learning embeddings have consistently shown superior performance over handcrafted acoustic features in speech emotion recognition (SER). However, unlike acoustic features with clear physical meaning, these embeddings lack clear interpretability. Explaining these embeddings is crucial for building trust in healthcare and security applications and advancing the scientific understanding of the acoustic information that is encoded in them. This paper proposes a modified probing approach to explain deep learning embeddings in the SER space. We predict interpretable acoustic feature"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.09511","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.09511/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.09511","created_at":"2026-07-05T09:07:10.689946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.09511v1","created_at":"2026-07-05T09:07:10.689946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.09511","created_at":"2026-07-05T09:07:10.689946+00:00"},{"alias_kind":"pith_short_12","alias_value":"4BYJSDQEQA24","created_at":"2026-07-05T09:07:10.689946+00:00"},{"alias_kind":"pith_short_16","alias_value":"4BYJSDQEQA24YCL7","created_at":"2026-07-05T09:07:10.689946+00:00"},{"alias_kind":"pith_short_8","alias_value":"4BYJSDQE","created_at":"2026-07-05T09:07:10.689946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.17542","citing_title":"Probing for Phonology in Self-Supervised Speech Representations: A Case Study on Accent Perception","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL","json":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL.json","graph_json":"https://pith.science/api/pith-number/4BYJSDQEQA24YCL7YQA3THMDAL/graph.json","events_json":"https://pith.science/api/pith-number/4BYJSDQEQA24YCL7YQA3THMDAL/events.json","paper":"https://pith.science/paper/4BYJSDQE"},"agent_actions":{"view_html":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL","download_json":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL.json","view_paper":"https://pith.science/paper/4BYJSDQE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.09511&json=true","fetch_graph":"https://pith.science/api/pith-number/4BYJSDQEQA24YCL7YQA3THMDAL/graph.json","fetch_events":"https://pith.science/api/pith-number/4BYJSDQEQA24YCL7YQA3THMDAL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL/action/storage_attestation","attest_author":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL/action/author_attestation","sign_citation":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL/action/citation_signature","submit_replication":"https://pith.science/pith/4BYJSDQEQA24YCL7YQA3THMDAL/action/replication_record"}},"created_at":"2026-07-05T09:07:10.689946+00:00","updated_at":"2026-07-05T09:07:10.689946+00:00"}