{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HCNYNAYMROUGPMY64DJIYTTHX7","short_pith_number":"pith:HCNYNAYM","schema_version":"1.0","canonical_sha256":"389b86830c8ba867b31ee0d28c4e67bfd438cd4f7f1b1e06153091387a915dcf","source":{"kind":"arxiv","id":"2502.03979","version":2},"attestation_state":"computed","paper":{"title":"Towards Unified Music Emotion Recognition across Dimensional and Categorical Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Dorien Herremans, Jaeyong Kang","submitted_at":"2025-02-06T11:20:22Z","abstract_excerpt":"One of the most significant challenges in Music Emotion Recognition (MER) comes from the fact that emotion labels can be heterogeneous across datasets with regard to the emotion representation, including categorical (e.g., happy, sad) versus dimensional labels (e.g., valence-arousal). In this paper, we present a unified multitask learning framework that combines these two types of labels and is thus able to be trained on multiple datasets. This framework uses an effective input representation that combines musical features (i.e., key and chords) and MERT embeddings. Moreover, knowledge distill"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03979","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2025-02-06T11:20:22Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"ce969bd2ef4ba31c1cd82bc9014607d2fced1b460ec6b0c36fca18401c696651","abstract_canon_sha256":"d72810bff882b3dae3264257b39f794eb2f13781c74ef4608ee69e208e3f0904"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:35.851386Z","signature_b64":"Qm4vqMQpDtkele1kNc3ewHsi8Ups7HGcQX5YJ9hSdUjov1jrRTuZJFUo+RBwb5S0hEec4lX8lWZMrI8m+rAFDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"389b86830c8ba867b31ee0d28c4e67bfd438cd4f7f1b1e06153091387a915dcf","last_reissued_at":"2026-07-05T10:47:35.850925Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:35.850925Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Unified Music Emotion Recognition across Dimensional and Categorical Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Dorien Herremans, Jaeyong Kang","submitted_at":"2025-02-06T11:20:22Z","abstract_excerpt":"One of the most significant challenges in Music Emotion Recognition (MER) comes from the fact that emotion labels can be heterogeneous across datasets with regard to the emotion representation, including categorical (e.g., happy, sad) versus dimensional labels (e.g., valence-arousal). In this paper, we present a unified multitask learning framework that combines these two types of labels and is thus able to be trained on multiple datasets. This framework uses an effective input representation that combines musical features (i.e., key and chords) and MERT embeddings. Moreover, knowledge distill"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03979","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03979/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03979","created_at":"2026-07-05T10:47:35.850983+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03979v2","created_at":"2026-07-05T10:47:35.850983+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03979","created_at":"2026-07-05T10:47:35.850983+00:00"},{"alias_kind":"pith_short_12","alias_value":"HCNYNAYMROUG","created_at":"2026-07-05T10:47:35.850983+00:00"},{"alias_kind":"pith_short_16","alias_value":"HCNYNAYMROUGPMY6","created_at":"2026-07-05T10:47:35.850983+00:00"},{"alias_kind":"pith_short_8","alias_value":"HCNYNAYM","created_at":"2026-07-05T10:47:35.850983+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24123","citing_title":"Aligning MusicLLM with Emotion using Instruction Tuning and Feedback-Driven Alignment","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28810","citing_title":"Affective Music Recommendation: A Rollout-Based World Model for Offline Preference Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23511","citing_title":"MECAT: A Multi-Experts Constructed Benchmark for Fine-Grained Audio Understanding Tasks","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08175","citing_title":"KARMA-MV: A Benchmark for Causal Question Answering on Music Videos","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7","json":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7.json","graph_json":"https://pith.science/api/pith-number/HCNYNAYMROUGPMY64DJIYTTHX7/graph.json","events_json":"https://pith.science/api/pith-number/HCNYNAYMROUGPMY64DJIYTTHX7/events.json","paper":"https://pith.science/paper/HCNYNAYM"},"agent_actions":{"view_html":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7","download_json":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7.json","view_paper":"https://pith.science/paper/HCNYNAYM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03979&json=true","fetch_graph":"https://pith.science/api/pith-number/HCNYNAYMROUGPMY64DJIYTTHX7/graph.json","fetch_events":"https://pith.science/api/pith-number/HCNYNAYMROUGPMY64DJIYTTHX7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7/action/storage_attestation","attest_author":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7/action/author_attestation","sign_citation":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7/action/citation_signature","submit_replication":"https://pith.science/pith/HCNYNAYMROUGPMY64DJIYTTHX7/action/replication_record"}},"created_at":"2026-07-05T10:47:35.850983+00:00","updated_at":"2026-07-05T10:47:35.850983+00:00"}