{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JB5NOWYFYKWLN46JRAEGS6NRWN","short_pith_number":"pith:JB5NOWYF","schema_version":"1.0","canonical_sha256":"487ad75b05c2acb6f3c988086979b1b373a9aa9fcf5bdba06c26b3da09952921","source":{"kind":"arxiv","id":"2501.01108","version":2},"attestation_state":"computed","paper":{"title":"MuQ: Self-Supervised Music Representation Learning with Mel Residual Vector Quantization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haina Zhu, Hangting Chen, Jianwei Yu, Rongzhi Gu, Wei Tan, Xie Chen, Yi Luo, Yizhi Zhou, Ziyang Ma","submitted_at":"2025-01-02T07:08:29Z","abstract_excerpt":"Recent years have witnessed the success of foundation models pre-trained with self-supervised learning (SSL) in various music informatics understanding tasks, including music tagging, instrument classification, key detection, and more. In this paper, we propose a self-supervised music representation learning model for music understanding. Distinguished from previous studies adopting random projection or existing neural codec, the proposed model, named MuQ, is trained to predict tokens generated by Mel Residual Vector Quantization (Mel-RVQ). Our Mel-RVQ utilizes residual linear projection struc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.01108","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2025-01-02T07:08:29Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","eess.AS"],"title_canon_sha256":"e8bd2c8b1b78a1c36bd2a330c60ad7b99168ab7687a5034b4dbeb741331de93e","abstract_canon_sha256":"971abc97cd4430e698f96109e65210e2c86e7c4cc57fae140dc41eac91eaf239"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:56:39.033596Z","signature_b64":"fGs/LuvvXM0yA20/xcUipytsdjjCxHme4nwvdyKilUWJtXcYN3qBE2YYumgk61ZtzcHstkz+F3xJnhwnkUktAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"487ad75b05c2acb6f3c988086979b1b373a9aa9fcf5bdba06c26b3da09952921","last_reissued_at":"2026-07-05T09:56:39.033076Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:56:39.033076Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MuQ: Self-Supervised Music Representation Learning with Mel Residual Vector Quantization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haina Zhu, Hangting Chen, Jianwei Yu, Rongzhi Gu, Wei Tan, Xie Chen, Yi Luo, Yizhi Zhou, Ziyang Ma","submitted_at":"2025-01-02T07:08:29Z","abstract_excerpt":"Recent years have witnessed the success of foundation models pre-trained with self-supervised learning (SSL) in various music informatics understanding tasks, including music tagging, instrument classification, key detection, and more. In this paper, we propose a self-supervised music representation learning model for music understanding. Distinguished from previous studies adopting random projection or existing neural codec, the proposed model, named MuQ, is trained to predict tokens generated by Mel Residual Vector Quantization (Mel-RVQ). Our Mel-RVQ utilizes residual linear projection struc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.01108","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.01108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.01108","created_at":"2026-07-05T09:56:39.033141+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.01108v2","created_at":"2026-07-05T09:56:39.033141+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.01108","created_at":"2026-07-05T09:56:39.033141+00:00"},{"alias_kind":"pith_short_12","alias_value":"JB5NOWYFYKWL","created_at":"2026-07-05T09:56:39.033141+00:00"},{"alias_kind":"pith_short_16","alias_value":"JB5NOWYFYKWLN46J","created_at":"2026-07-05T09:56:39.033141+00:00"},{"alias_kind":"pith_short_8","alias_value":"JB5NOWYF","created_at":"2026-07-05T09:56:39.033141+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06929","citing_title":"MADB: A Large-Scale Music Aesthetics Dataset with Professional and Multi-Dimensional Annotations","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07387","citing_title":"Making the Most of Limited Data: Score-Aware Training for Text-to-Music Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06615","citing_title":"FIGMA: Towards FIne-Grained Music retrievAl","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01677","citing_title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11910","citing_title":"TADA! Tuning Audio Diffusion Models through Activation Steering","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03190","citing_title":"Expectation and Acoustic Neural Network Representations Enhance Music Identification from Brain Activity","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17414","citing_title":"S2Accompanist: A Semantic-Aware and Structure-Guided Diffusion Model for Music Accompaniment Generation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02797","citing_title":"SongFormer: Scaling Music Structure Analysis with Heterogeneous Supervision","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20847","citing_title":"Revisiting Content-Based Music Recommendation: Efficient Feature Aggregation from Large-Scale Music Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23077","citing_title":"Adopting State-of-the-Art Pretrained Audio Representations for Music Recommender Systems","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07090","citing_title":"Leveraging Artist Catalogs for Cold-Start Music Recommendation","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2509.17765","citing_title":"Qwen3-Omni Technical Report","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN","json":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN.json","graph_json":"https://pith.science/api/pith-number/JB5NOWYFYKWLN46JRAEGS6NRWN/graph.json","events_json":"https://pith.science/api/pith-number/JB5NOWYFYKWLN46JRAEGS6NRWN/events.json","paper":"https://pith.science/paper/JB5NOWYF"},"agent_actions":{"view_html":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN","download_json":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN.json","view_paper":"https://pith.science/paper/JB5NOWYF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.01108&json=true","fetch_graph":"https://pith.science/api/pith-number/JB5NOWYFYKWLN46JRAEGS6NRWN/graph.json","fetch_events":"https://pith.science/api/pith-number/JB5NOWYFYKWLN46JRAEGS6NRWN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN/action/storage_attestation","attest_author":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN/action/author_attestation","sign_citation":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN/action/citation_signature","submit_replication":"https://pith.science/pith/JB5NOWYFYKWLN46JRAEGS6NRWN/action/replication_record"}},"created_at":"2026-07-05T09:56:39.033141+00:00","updated_at":"2026-07-05T09:56:39.033141+00:00"}