{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LO2NQBDN7AKYBMNHWGDT4T2JNL","short_pith_number":"pith:LO2NQBDN","schema_version":"1.0","canonical_sha256":"5bb4d8046df81580b1a7b1873e4f496af8da3ca277a9f9fdd133ecfc801d30ca","source":{"kind":"arxiv","id":"2402.07596","version":2},"attestation_state":"computed","paper":{"title":"Sheet Music Transformer: End-To-End Optical Music Recognition Beyond Monophonic Transcription","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Antonio R\\'ios-Vila, Jorge Calvo-Zaragoza, Thierry Paquet","submitted_at":"2024-02-12T11:52:21Z","abstract_excerpt":"State-of-the-art end-to-end Optical Music Recognition (OMR) has, to date, primarily been carried out using monophonic transcription techniques to handle complex score layouts, such as polyphony, often by resorting to simplifications or specific adaptations. Despite their efficacy, these approaches imply challenges related to scalability and limitations. This paper presents the Sheet Music Transformer, the first end-to-end OMR model designed to transcribe complex musical scores without relying solely on monophonic strategies. Our model employs a Transformer-based image-to-sequence framework tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.07596","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-12T11:52:21Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"2d5052b88788c75b62bcb5cb44977f040800b3c11868de2725d4ac658e94d051","abstract_canon_sha256":"2af5fed1791f136fe187ca55b9f1c1e167f14cb114532effe0b790ac380ac101"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:13:12.232847Z","signature_b64":"mS/klSc0UIJDVpLWuysaG7m8aq3Yz74FU9aAcMMwrX6NrSN6uf8lYn7nFHgdO4k7iy2M6WxIjxNqeYpSF41iCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5bb4d8046df81580b1a7b1873e4f496af8da3ca277a9f9fdd133ecfc801d30ca","last_reissued_at":"2026-07-05T08:13:12.232356Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:13:12.232356Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sheet Music Transformer: End-To-End Optical Music Recognition Beyond Monophonic Transcription","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Antonio R\\'ios-Vila, Jorge Calvo-Zaragoza, Thierry Paquet","submitted_at":"2024-02-12T11:52:21Z","abstract_excerpt":"State-of-the-art end-to-end Optical Music Recognition (OMR) has, to date, primarily been carried out using monophonic transcription techniques to handle complex score layouts, such as polyphony, often by resorting to simplifications or specific adaptations. Despite their efficacy, these approaches imply challenges related to scalability and limitations. This paper presents the Sheet Music Transformer, the first end-to-end OMR model designed to transcribe complex musical scores without relying solely on monophonic strategies. Our model employs a Transformer-based image-to-sequence framework tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07596","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.07596/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.07596","created_at":"2026-07-05T08:13:12.232417+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.07596v2","created_at":"2026-07-05T08:13:12.232417+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07596","created_at":"2026-07-05T08:13:12.232417+00:00"},{"alias_kind":"pith_short_12","alias_value":"LO2NQBDN7AKY","created_at":"2026-07-05T08:13:12.232417+00:00"},{"alias_kind":"pith_short_16","alias_value":"LO2NQBDN7AKYBMNH","created_at":"2026-07-05T08:13:12.232417+00:00"},{"alias_kind":"pith_short_8","alias_value":"LO2NQBDN","created_at":"2026-07-05T08:13:12.232417+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09479","citing_title":"Optical Music Recognition for Real-World Manuscripts with Synthetic Data","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2409.01704","citing_title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16446","citing_title":"A High-Accuracy Optical Music Recognition Method Based on Bottleneck Residual Convolutions","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL","json":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL.json","graph_json":"https://pith.science/api/pith-number/LO2NQBDN7AKYBMNHWGDT4T2JNL/graph.json","events_json":"https://pith.science/api/pith-number/LO2NQBDN7AKYBMNHWGDT4T2JNL/events.json","paper":"https://pith.science/paper/LO2NQBDN"},"agent_actions":{"view_html":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL","download_json":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL.json","view_paper":"https://pith.science/paper/LO2NQBDN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.07596&json=true","fetch_graph":"https://pith.science/api/pith-number/LO2NQBDN7AKYBMNHWGDT4T2JNL/graph.json","fetch_events":"https://pith.science/api/pith-number/LO2NQBDN7AKYBMNHWGDT4T2JNL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL/action/storage_attestation","attest_author":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL/action/author_attestation","sign_citation":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL/action/citation_signature","submit_replication":"https://pith.science/pith/LO2NQBDN7AKYBMNHWGDT4T2JNL/action/replication_record"}},"created_at":"2026-07-05T08:13:12.232417+00:00","updated_at":"2026-07-05T08:13:12.232417+00:00"}