{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NVQOYLM365SXRMAJPEB2XS442Y","short_pith_number":"pith:NVQOYLM3","schema_version":"1.0","canonical_sha256":"6d60ec2d9bf76578b0097903abcb9cd61e08c9646b8d393cdbc2072a085085b9","source":{"kind":"arxiv","id":"2409.04702","version":1},"attestation_state":"computed","paper":{"title":"Mel-RoFormer for Vocal Separation and Vocal Melody Transcription","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Jitong Chen, Ju-Chiang Wang, Wei-Tsung Lu","submitted_at":"2024-09-07T03:55:02Z","abstract_excerpt":"Developing a versatile deep neural network to model music audio is crucial in MIR. This task is challenging due to the intricate spectral variations inherent in music signals, which convey melody, harmonics, and timbres of diverse instruments. In this paper, we introduce Mel-RoFormer, a spectrogram-based model featuring two key designs: a novel Mel-band Projection module at the front-end to enhance the model's capability to capture informative features across multiple frequency bands, and interleaved RoPE Transformers to explicitly model the frequency and time dimensions as two separate sequen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.04702","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-09-07T03:55:02Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"2ee8c5cbb347a06dc201766cc31339c1ed9fad0187a25a4eff7ed02af89a3016","abstract_canon_sha256":"89b624d37496e30c5ff70dff418caa8c46fae97c615d9bf9c3347aecdc5f8412"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:04:16.533801Z","signature_b64":"t4TSga+vBCPN/f4aeVNKnXrsTeGs8/4kqderC1Takqcbg/X8sLgI1wU2WI/SYPvBgva47yl1PshgQ96PMav5Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d60ec2d9bf76578b0097903abcb9cd61e08c9646b8d393cdbc2072a085085b9","last_reissued_at":"2026-07-05T09:04:16.533394Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:04:16.533394Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mel-RoFormer for Vocal Separation and Vocal Melody Transcription","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Jitong Chen, Ju-Chiang Wang, Wei-Tsung Lu","submitted_at":"2024-09-07T03:55:02Z","abstract_excerpt":"Developing a versatile deep neural network to model music audio is crucial in MIR. This task is challenging due to the intricate spectral variations inherent in music signals, which convey melody, harmonics, and timbres of diverse instruments. In this paper, we introduce Mel-RoFormer, a spectrogram-based model featuring two key designs: a novel Mel-band Projection module at the front-end to enhance the model's capability to capture informative features across multiple frequency bands, and interleaved RoPE Transformers to explicitly model the frequency and time dimensions as two separate sequen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.04702","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.04702/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.04702","created_at":"2026-07-05T09:04:16.533452+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.04702v1","created_at":"2026-07-05T09:04:16.533452+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.04702","created_at":"2026-07-05T09:04:16.533452+00:00"},{"alias_kind":"pith_short_12","alias_value":"NVQOYLM365SX","created_at":"2026-07-05T09:04:16.533452+00:00"},{"alias_kind":"pith_short_16","alias_value":"NVQOYLM365SXRMAJ","created_at":"2026-07-05T09:04:16.533452+00:00"},{"alias_kind":"pith_short_8","alias_value":"NVQOYLM3","created_at":"2026-07-05T09:04:16.533452+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03168","citing_title":"JAVEDIT: Joint Audio-Visual Instruction-Guided Video Editing with Agentic Data Curation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08729","citing_title":"Unison: Harmonizing Motion, Speech, and Sound for Human-Centric Audio-Video Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08729","citing_title":"Unison: Harmonizing Motion, Speech, and Sound for Human-Centric Audio-Video Generation","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y","json":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y.json","graph_json":"https://pith.science/api/pith-number/NVQOYLM365SXRMAJPEB2XS442Y/graph.json","events_json":"https://pith.science/api/pith-number/NVQOYLM365SXRMAJPEB2XS442Y/events.json","paper":"https://pith.science/paper/NVQOYLM3"},"agent_actions":{"view_html":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y","download_json":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y.json","view_paper":"https://pith.science/paper/NVQOYLM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.04702&json=true","fetch_graph":"https://pith.science/api/pith-number/NVQOYLM365SXRMAJPEB2XS442Y/graph.json","fetch_events":"https://pith.science/api/pith-number/NVQOYLM365SXRMAJPEB2XS442Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y/action/storage_attestation","attest_author":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y/action/author_attestation","sign_citation":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y/action/citation_signature","submit_replication":"https://pith.science/pith/NVQOYLM365SXRMAJPEB2XS442Y/action/replication_record"}},"created_at":"2026-07-05T09:04:16.533452+00:00","updated_at":"2026-07-05T09:04:16.533452+00:00"}