{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:ZHK7U54AVFCOJARGVT2R42HUXD","short_pith_number":"pith:ZHK7U54A","schema_version":"1.0","canonical_sha256":"c9d5fa7780a944e48226acf51e68f4b8ceefb1dd3ac4debadb73a6b49ab2d7bc","source":{"kind":"arxiv","id":"2607.06015","version":1},"attestation_state":"computed","paper":{"title":"Music I Care About: Automated Multimodal Benchmarking of LLM Music Perception Skills on (Almost) Any Music","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SD","authors_text":"Jan Haji\\v{c} jr, Katia Vendrame, Tom\\'a\\v{s} Sourada","submitted_at":"2026-07-07T08:57:34Z","abstract_excerpt":"Music represents a cornerstone of human culture, existing digitally across diverse modalities, including audio, symbolic encodings (e.g., MIDI, MusicXML), and sheet music. Despite the advancement of Multimodal Large Language Models (MLLMs), current music benchmarks face three major limitations. First, large static benchmarks are resource-intensive to evaluate, and it remains unclear how their results transfer to diverse kinds of music beyond those included in the benchmark. Second, benchmarks claiming to measure \"music understanding\" often fail to require music perception. Third, they do not s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.06015","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2026-07-07T08:57:34Z","cross_cats_sorted":[],"title_canon_sha256":"5f21c9a9a7d4639325bb6b63d6aaf40221544da35a18d7fff05c0e087850c4ed","abstract_canon_sha256":"f7b4c204f40bc6dd719b204b0bf95ec7a91c921c2a8d481827d82b5a8457bd30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-08T01:18:54.023224Z","signature_b64":"SJGoPyZvu9BVAsu7siI9A7mCugwhATC1+REPaRBt9wAOiSEgF3emuN5Ggt3f0OP0hJNzTlv9oJ93EnfUDso8Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9d5fa7780a944e48226acf51e68f4b8ceefb1dd3ac4debadb73a6b49ab2d7bc","last_reissued_at":"2026-07-08T01:18:54.022807Z","signature_status":"signed_v1","first_computed_at":"2026-07-08T01:18:54.022807Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Music I Care About: Automated Multimodal Benchmarking of LLM Music Perception Skills on (Almost) Any Music","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SD","authors_text":"Jan Haji\\v{c} jr, Katia Vendrame, Tom\\'a\\v{s} Sourada","submitted_at":"2026-07-07T08:57:34Z","abstract_excerpt":"Music represents a cornerstone of human culture, existing digitally across diverse modalities, including audio, symbolic encodings (e.g., MIDI, MusicXML), and sheet music. Despite the advancement of Multimodal Large Language Models (MLLMs), current music benchmarks face three major limitations. First, large static benchmarks are resource-intensive to evaluate, and it remains unclear how their results transfer to diverse kinds of music beyond those included in the benchmark. Second, benchmarks claiming to measure \"music understanding\" often fail to require music perception. Third, they do not s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.06015","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.06015/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.06015","created_at":"2026-07-08T01:18:54.022866+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.06015v1","created_at":"2026-07-08T01:18:54.022866+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.06015","created_at":"2026-07-08T01:18:54.022866+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZHK7U54AVFCO","created_at":"2026-07-08T01:18:54.022866+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZHK7U54AVFCOJARG","created_at":"2026-07-08T01:18:54.022866+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZHK7U54A","created_at":"2026-07-08T01:18:54.022866+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06015","citing_title":"Music I Care About: Automated Multimodal Benchmarking of LLM Music Perception Skills on (Almost) Any Music","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD","json":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD.json","graph_json":"https://pith.science/api/pith-number/ZHK7U54AVFCOJARGVT2R42HUXD/graph.json","events_json":"https://pith.science/api/pith-number/ZHK7U54AVFCOJARGVT2R42HUXD/events.json","paper":"https://pith.science/paper/ZHK7U54A"},"agent_actions":{"view_html":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD","download_json":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD.json","view_paper":"https://pith.science/paper/ZHK7U54A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.06015&json=true","fetch_graph":"https://pith.science/api/pith-number/ZHK7U54AVFCOJARGVT2R42HUXD/graph.json","fetch_events":"https://pith.science/api/pith-number/ZHK7U54AVFCOJARGVT2R42HUXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD/action/storage_attestation","attest_author":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD/action/author_attestation","sign_citation":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD/action/citation_signature","submit_replication":"https://pith.science/pith/ZHK7U54AVFCOJARGVT2R42HUXD/action/replication_record"}},"created_at":"2026-07-08T01:18:54.022866+00:00","updated_at":"2026-07-08T01:18:54.022866+00:00"}