{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LXJV4GUJXG7OTITH7PRLEI2IVO","short_pith_number":"pith:LXJV4GUJ","schema_version":"1.0","canonical_sha256":"5dd35e1a89b9bee9a267fbe2b22348ab8ac7a614eb5242fc98c6149da8803d4d","source":{"kind":"arxiv","id":"2504.18799","version":1},"attestation_state":"computed","paper":{"title":"A Survey on Multimodal Music Emotion Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Aditya Joshi, Erik Meijering, Rashini Liyanarachchi","submitted_at":"2025-04-26T05:11:04Z","abstract_excerpt":"Multimodal music emotion recognition (MMER) is an emerging discipline in music information retrieval that has experienced a surge in interest in recent years. This survey provides a comprehensive overview of the current state-of-the-art in MMER. Discussing the different approaches and techniques used in this field, the paper introduces a four-stage MMER framework, including multimodal data selection, feature extraction, feature processing, and final emotion prediction. The survey further reveals significant advancements in deep learning methods and the increasing importance of feature fusion t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.18799","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MM","submitted_at":"2025-04-26T05:11:04Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"ae5fad4091098de1a77ce419fbb3d16f6ca3d38ff5f393a1d5189a087c606600","abstract_canon_sha256":"ace6126e00c11bbc7923cad60ebaef89dbae60369f677591b2003ff01e87fb77"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:21.533525Z","signature_b64":"pP4whn37prP+R3Y5iBgjhEqsyfOEda3wE4HxRCHPvIR72wUfss8yxhnJzuArHT+HDrWKeEnEgGxQCnKKnxtQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5dd35e1a89b9bee9a267fbe2b22348ab8ac7a614eb5242fc98c6149da8803d4d","last_reissued_at":"2026-07-05T10:54:21.532993Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:21.532993Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Multimodal Music Emotion Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Aditya Joshi, Erik Meijering, Rashini Liyanarachchi","submitted_at":"2025-04-26T05:11:04Z","abstract_excerpt":"Multimodal music emotion recognition (MMER) is an emerging discipline in music information retrieval that has experienced a surge in interest in recent years. This survey provides a comprehensive overview of the current state-of-the-art in MMER. Discussing the different approaches and techniques used in this field, the paper introduces a four-stage MMER framework, including multimodal data selection, feature extraction, feature processing, and final emotion prediction. The survey further reveals significant advancements in deep learning methods and the increasing importance of feature fusion t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.18799","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.18799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.18799","created_at":"2026-07-05T10:54:21.533049+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.18799v1","created_at":"2026-07-05T10:54:21.533049+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.18799","created_at":"2026-07-05T10:54:21.533049+00:00"},{"alias_kind":"pith_short_12","alias_value":"LXJV4GUJXG7O","created_at":"2026-07-05T10:54:21.533049+00:00"},{"alias_kind":"pith_short_16","alias_value":"LXJV4GUJXG7OTITH","created_at":"2026-07-05T10:54:21.533049+00:00"},{"alias_kind":"pith_short_8","alias_value":"LXJV4GUJ","created_at":"2026-07-05T10:54:21.533049+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29273","citing_title":"A Hybrid Framework for Song Lyric Annotation Based on Human-LLM Alignment","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO","json":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO.json","graph_json":"https://pith.science/api/pith-number/LXJV4GUJXG7OTITH7PRLEI2IVO/graph.json","events_json":"https://pith.science/api/pith-number/LXJV4GUJXG7OTITH7PRLEI2IVO/events.json","paper":"https://pith.science/paper/LXJV4GUJ"},"agent_actions":{"view_html":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO","download_json":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO.json","view_paper":"https://pith.science/paper/LXJV4GUJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.18799&json=true","fetch_graph":"https://pith.science/api/pith-number/LXJV4GUJXG7OTITH7PRLEI2IVO/graph.json","fetch_events":"https://pith.science/api/pith-number/LXJV4GUJXG7OTITH7PRLEI2IVO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO/action/storage_attestation","attest_author":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO/action/author_attestation","sign_citation":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO/action/citation_signature","submit_replication":"https://pith.science/pith/LXJV4GUJXG7OTITH7PRLEI2IVO/action/replication_record"}},"created_at":"2026-07-05T10:54:21.533049+00:00","updated_at":"2026-07-05T10:54:21.533049+00:00"}