{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KQHP6KTN5K57GLFOICZPWOF4KI","short_pith_number":"pith:KQHP6KTN","schema_version":"1.0","canonical_sha256":"540eff2a6deabbf32cae40b2fb38bc523feedd9c50d90367a2ff0553bd51e327","source":{"kind":"arxiv","id":"2501.10666","version":1},"attestation_state":"computed","paper":{"title":"Speech Emotion Detection Based on MFCC and CNN-LSTM Architecture","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Qianhe Ouyang","submitted_at":"2025-01-18T06:15:54Z","abstract_excerpt":"Emotion detection techniques have been applied to multiple cases mainly from facial image features and vocal audio features, of which the latter aspect is disputed yet not only due to the complexity of speech audio processing but also the difficulties of extracting appropriate features. Part of the SAVEE and RAVDESS datasets are selected and combined as the dataset, containing seven sorts of common emotions (i.e. happy, neutral, sad, anger, disgust, fear, and surprise) and thousands of samples. Based on the Librosa package, this paper processes the initial audio input into waveplot and spectru"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.10666","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2025-01-18T06:15:54Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"0216e9686c2055a554a12557bc991d4f702a890be25cd71f942980652f27397b","abstract_canon_sha256":"2f11555fa73225029fd2635ae6fec59c1204f07e18e6500dfba0a3cc57be4fa1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:31.612479Z","signature_b64":"s3ejsLaJwchoW00EvfuUXs5sitSmmPhzZzai8ukgTaEkj36WmnlfbypcZWksMAVG04DNdzhsriSJZN5GjToXBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"540eff2a6deabbf32cae40b2fb38bc523feedd9c50d90367a2ff0553bd51e327","last_reissued_at":"2026-07-05T10:02:31.612007Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:31.612007Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speech Emotion Detection Based on MFCC and CNN-LSTM Architecture","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Qianhe Ouyang","submitted_at":"2025-01-18T06:15:54Z","abstract_excerpt":"Emotion detection techniques have been applied to multiple cases mainly from facial image features and vocal audio features, of which the latter aspect is disputed yet not only due to the complexity of speech audio processing but also the difficulties of extracting appropriate features. Part of the SAVEE and RAVDESS datasets are selected and combined as the dataset, containing seven sorts of common emotions (i.e. happy, neutral, sad, anger, disgust, fear, and surprise) and thousands of samples. Based on the Librosa package, this paper processes the initial audio input into waveplot and spectru"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.10666","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.10666/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.10666","created_at":"2026-07-05T10:02:31.612076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.10666v1","created_at":"2026-07-05T10:02:31.612076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.10666","created_at":"2026-07-05T10:02:31.612076+00:00"},{"alias_kind":"pith_short_12","alias_value":"KQHP6KTN5K57","created_at":"2026-07-05T10:02:31.612076+00:00"},{"alias_kind":"pith_short_16","alias_value":"KQHP6KTN5K57GLFO","created_at":"2026-07-05T10:02:31.612076+00:00"},{"alias_kind":"pith_short_8","alias_value":"KQHP6KTN","created_at":"2026-07-05T10:02:31.612076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26107","citing_title":"Low Resource Multimodal Translation of Nepali Spoken Words into Emotion-Conditioned Sign Language Avatars","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI","json":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI.json","graph_json":"https://pith.science/api/pith-number/KQHP6KTN5K57GLFOICZPWOF4KI/graph.json","events_json":"https://pith.science/api/pith-number/KQHP6KTN5K57GLFOICZPWOF4KI/events.json","paper":"https://pith.science/paper/KQHP6KTN"},"agent_actions":{"view_html":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI","download_json":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI.json","view_paper":"https://pith.science/paper/KQHP6KTN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.10666&json=true","fetch_graph":"https://pith.science/api/pith-number/KQHP6KTN5K57GLFOICZPWOF4KI/graph.json","fetch_events":"https://pith.science/api/pith-number/KQHP6KTN5K57GLFOICZPWOF4KI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI/action/storage_attestation","attest_author":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI/action/author_attestation","sign_citation":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI/action/citation_signature","submit_replication":"https://pith.science/pith/KQHP6KTN5K57GLFOICZPWOF4KI/action/replication_record"}},"created_at":"2026-07-05T10:02:31.612076+00:00","updated_at":"2026-07-05T10:02:31.612076+00:00"}