{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TIQXH57AL5SXEPWUCHUERMWHTI","short_pith_number":"pith:TIQXH57A","schema_version":"1.0","canonical_sha256":"9a2173f7e05f65723ed411e848b2c79a2e4c760dfe878d3086035cc7ae8b5a13","source":{"kind":"arxiv","id":"2506.23009","version":3},"attestation_state":"computed","paper":{"title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changyou Chen, Chenguang Wang, Jian Chen, Jiayu Qin, Ming Li, Penghang Liu, Ruiyi Zhang, Tengwei Song, Wei Wang, Wenye Ma","submitted_at":"2025-06-28T20:46:47Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved remarkable visual reasoning abilities in natural images, text-rich documents, and graphic designs. However, their ability to interpret music sheets remains underexplored. To bridge this gap, we introduce MusiXQA, the first comprehensive dataset for evaluating and advancing MLLMs in music sheet understanding. MusiXQA features high-quality synthetic music sheets generated via MusiXTeX, with structured annotations covering note pitch and duration, chords, clefs, key/time signatures, and text, enabling diverse visual QA tasks. Through extensiv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.23009","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-28T20:46:47Z","cross_cats_sorted":[],"title_canon_sha256":"45d158b453104489dd7700e40ac8266d10cc5b75abf8dfa0bb84f8661929b01a","abstract_canon_sha256":"aeca31c303cfa7e3c01bf3b923268f3978b2022c4742cd0654a38d4b6b98724e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:48.314907Z","signature_b64":"0YC2Z8UpgoDTo+OGxOV7/2YBBih4ZbbrRuH9Fsak57pvLOs5ItmN3zyf/QDC1k8t3A4rrnCkCg5JsFN4PPEcAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a2173f7e05f65723ed411e848b2c79a2e4c760dfe878d3086035cc7ae8b5a13","last_reissued_at":"2026-07-05T11:53:48.314241Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:48.314241Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changyou Chen, Chenguang Wang, Jian Chen, Jiayu Qin, Ming Li, Penghang Liu, Ruiyi Zhang, Tengwei Song, Wei Wang, Wenye Ma","submitted_at":"2025-06-28T20:46:47Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved remarkable visual reasoning abilities in natural images, text-rich documents, and graphic designs. However, their ability to interpret music sheets remains underexplored. To bridge this gap, we introduce MusiXQA, the first comprehensive dataset for evaluating and advancing MLLMs in music sheet understanding. MusiXQA features high-quality synthetic music sheets generated via MusiXTeX, with structured annotations covering note pitch and duration, chords, clefs, key/time signatures, and text, enabling diverse visual QA tasks. Through extensiv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.23009","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.23009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.23009","created_at":"2026-07-05T11:53:48.314330+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.23009v3","created_at":"2026-07-05T11:53:48.314330+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.23009","created_at":"2026-07-05T11:53:48.314330+00:00"},{"alias_kind":"pith_short_12","alias_value":"TIQXH57AL5SX","created_at":"2026-07-05T11:53:48.314330+00:00"},{"alias_kind":"pith_short_16","alias_value":"TIQXH57AL5SXEPWU","created_at":"2026-07-05T11:53:48.314330+00:00"},{"alias_kind":"pith_short_8","alias_value":"TIQXH57A","created_at":"2026-07-05T11:53:48.314330+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05769","citing_title":"LEGATO 2: Toward Multimodal Sheet Music Recognition and Understanding","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06015","citing_title":"Music I Care About: Automated Multimodal Benchmarking of LLM Music Perception Skills on (Almost) Any Music","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22255","citing_title":"Direct content-based retrieval from music scores images","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22255","citing_title":"Direct content-based retrieval from music scores images","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20719","citing_title":"ONOTE: Benchmarking Omnimodal Notation Processing for Expert-level Music Intelligence","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI","json":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI.json","graph_json":"https://pith.science/api/pith-number/TIQXH57AL5SXEPWUCHUERMWHTI/graph.json","events_json":"https://pith.science/api/pith-number/TIQXH57AL5SXEPWUCHUERMWHTI/events.json","paper":"https://pith.science/paper/TIQXH57A"},"agent_actions":{"view_html":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI","download_json":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI.json","view_paper":"https://pith.science/paper/TIQXH57A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.23009&json=true","fetch_graph":"https://pith.science/api/pith-number/TIQXH57AL5SXEPWUCHUERMWHTI/graph.json","fetch_events":"https://pith.science/api/pith-number/TIQXH57AL5SXEPWUCHUERMWHTI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI/action/storage_attestation","attest_author":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI/action/author_attestation","sign_citation":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI/action/citation_signature","submit_replication":"https://pith.science/pith/TIQXH57AL5SXEPWUCHUERMWHTI/action/replication_record"}},"created_at":"2026-07-05T11:53:48.314330+00:00","updated_at":"2026-07-05T11:53:48.314330+00:00"}