{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T5CCPS4V2V3XE6VQJC2EDTHCGW","short_pith_number":"pith:T5CCPS4V","schema_version":"1.0","canonical_sha256":"9f4427cb95d577727ab048b441cce2358071a6e83c62c2571a0af4992241d0c6","source":{"kind":"arxiv","id":"2408.01337","version":1},"attestation_state":"computed","paper":{"title":"MuChoMusic: Evaluating Music Understanding in Multimodal Audio-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Benno Weck, Dmitry Bogdanov, Elio Quinton, Emmanouil Benetos, George Fazekas, Ilaria Manco","submitted_at":"2024-08-02T15:34:05Z","abstract_excerpt":"Multimodal models that jointly process audio and language hold great promise in audio understanding and are increasingly being adopted in the music domain. By allowing users to query via text and obtain information about a given audio input, these models have the potential to enable a variety of music understanding tasks via language-based interfaces. However, their evaluation poses considerable challenges, and it remains unclear how to effectively assess their ability to correctly interpret music-related inputs with current methods. Motivated by this, we introduce MuChoMusic, a benchmark for "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.01337","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-08-02T15:34:05Z","cross_cats_sorted":["cs.CL","cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"8bef076194901b7b82b04cea56c8cbb52b97db77d80e4fb6747d2c5ea9654d77","abstract_canon_sha256":"8fa6344d4601dc495999431ea98ea06e422b6478b2d4b4d7eb28ff90f1a84f78"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:30.326796Z","signature_b64":"okTEo+lDXzZT7B200U2dzKi8ogHS62/vwnqWlcXJk3HQPmDcsCBT3FMxxRqi8zE/NQUnb32M/7fBRbahTfC3Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f4427cb95d577727ab048b441cce2358071a6e83c62c2571a0af4992241d0c6","last_reissued_at":"2026-07-05T08:51:30.326323Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:30.326323Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MuChoMusic: Evaluating Music Understanding in Multimodal Audio-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Benno Weck, Dmitry Bogdanov, Elio Quinton, Emmanouil Benetos, George Fazekas, Ilaria Manco","submitted_at":"2024-08-02T15:34:05Z","abstract_excerpt":"Multimodal models that jointly process audio and language hold great promise in audio understanding and are increasingly being adopted in the music domain. By allowing users to query via text and obtain information about a given audio input, these models have the potential to enable a variety of music understanding tasks via language-based interfaces. However, their evaluation poses considerable challenges, and it remains unclear how to effectively assess their ability to correctly interpret music-related inputs with current methods. Motivated by this, we introduce MuChoMusic, a benchmark for "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.01337","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.01337/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.01337","created_at":"2026-07-05T08:51:30.326383+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.01337v1","created_at":"2026-07-05T08:51:30.326383+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.01337","created_at":"2026-07-05T08:51:30.326383+00:00"},{"alias_kind":"pith_short_12","alias_value":"T5CCPS4V2V3X","created_at":"2026-07-05T08:51:30.326383+00:00"},{"alias_kind":"pith_short_16","alias_value":"T5CCPS4V2V3XE6VQ","created_at":"2026-07-05T08:51:30.326383+00:00"},{"alias_kind":"pith_short_8","alias_value":"T5CCPS4V","created_at":"2026-07-05T08:51:30.326383+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06015","citing_title":"Music I Care About: Automated Multimodal Benchmarking of LLM Music Perception Skills on (Almost) Any Music","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31338","citing_title":"Beyond Binary Instrument QA: Probing Instrument Grounding in Music Audio-Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29300","citing_title":"MusTBENCH: Benchmarking and Advancing Temporal Grounding in Music LLMs","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20638","citing_title":"Music Audio-Visual Question Answering Requires Specialized Multimodal Designs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.08128","citing_title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16502","citing_title":"Topology-Aware Layer Pruning for Large Vision-Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW","json":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW.json","graph_json":"https://pith.science/api/pith-number/T5CCPS4V2V3XE6VQJC2EDTHCGW/graph.json","events_json":"https://pith.science/api/pith-number/T5CCPS4V2V3XE6VQJC2EDTHCGW/events.json","paper":"https://pith.science/paper/T5CCPS4V"},"agent_actions":{"view_html":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW","download_json":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW.json","view_paper":"https://pith.science/paper/T5CCPS4V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.01337&json=true","fetch_graph":"https://pith.science/api/pith-number/T5CCPS4V2V3XE6VQJC2EDTHCGW/graph.json","fetch_events":"https://pith.science/api/pith-number/T5CCPS4V2V3XE6VQJC2EDTHCGW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW/action/storage_attestation","attest_author":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW/action/author_attestation","sign_citation":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW/action/citation_signature","submit_replication":"https://pith.science/pith/T5CCPS4V2V3XE6VQJC2EDTHCGW/action/replication_record"}},"created_at":"2026-07-05T08:51:30.326383+00:00","updated_at":"2026-07-05T08:51:30.326383+00:00"}