{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6W27IB7SZIS2CCHSG4CINIDZLM","short_pith_number":"pith:6W27IB7S","schema_version":"1.0","canonical_sha256":"f5b5f407f2ca25a108f2370486a0795b05ed2174c02546a121ff91e9a6b77afd","source":{"kind":"arxiv","id":"2408.04737","version":1},"attestation_state":"computed","paper":{"title":"Quantifying the Corpus Bias Problem in Automatic Music Transcription Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Gerhard Widmer, Luk\\'a\\v{s} Samuel Mart\\'ak, Patricia Hu","submitted_at":"2024-08-08T19:40:28Z","abstract_excerpt":"Automatic Music Transcription (AMT) is the task of recognizing notes in audio recordings of music. The State-of-the-Art (SotA) benchmarks have been dominated by deep learning systems. Due to the scarcity of high quality data, they are usually trained and evaluated exclusively or predominantly on classical piano music. Unfortunately, that hinders our ability to understand how they generalize to other music. Previous works have revealed several aspects of memorization and overfitting in these systems. We identify two primary sources of distribution shift: the music, and the sound. Complementing "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.04737","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-08-08T19:40:28Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"b206cc88eeb9a8c9b1caa169eff393cb55bd1ab35919cc2ec1257acdc1147ab8","abstract_canon_sha256":"3165cff16a88c475cd2b311e0de784cf769b81e1d08580f08b3d92021ba91e9c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:53:42.899928Z","signature_b64":"nMQPKCjgpEmEFrEQfccjgw86elCWBSjLJXd6WvbcHpQgpIu4gD2BCIpI79LO2gYOoVznCTBqKqY6BP5wqES/Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5b5f407f2ca25a108f2370486a0795b05ed2174c02546a121ff91e9a6b77afd","last_reissued_at":"2026-07-05T08:53:42.899417Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:53:42.899417Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantifying the Corpus Bias Problem in Automatic Music Transcription Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Gerhard Widmer, Luk\\'a\\v{s} Samuel Mart\\'ak, Patricia Hu","submitted_at":"2024-08-08T19:40:28Z","abstract_excerpt":"Automatic Music Transcription (AMT) is the task of recognizing notes in audio recordings of music. The State-of-the-Art (SotA) benchmarks have been dominated by deep learning systems. Due to the scarcity of high quality data, they are usually trained and evaluated exclusively or predominantly on classical piano music. Unfortunately, that hinders our ability to understand how they generalize to other music. Previous works have revealed several aspects of memorization and overfitting in these systems. We identify two primary sources of distribution shift: the music, and the sound. Complementing "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.04737","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.04737/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.04737","created_at":"2026-07-05T08:53:42.899477+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.04737v1","created_at":"2026-07-05T08:53:42.899477+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.04737","created_at":"2026-07-05T08:53:42.899477+00:00"},{"alias_kind":"pith_short_12","alias_value":"6W27IB7SZIS2","created_at":"2026-07-05T08:53:42.899477+00:00"},{"alias_kind":"pith_short_16","alias_value":"6W27IB7SZIS2CCHS","created_at":"2026-07-05T08:53:42.899477+00:00"},{"alias_kind":"pith_short_8","alias_value":"6W27IB7S","created_at":"2026-07-05T08:53:42.899477+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM","json":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM.json","graph_json":"https://pith.science/api/pith-number/6W27IB7SZIS2CCHSG4CINIDZLM/graph.json","events_json":"https://pith.science/api/pith-number/6W27IB7SZIS2CCHSG4CINIDZLM/events.json","paper":"https://pith.science/paper/6W27IB7S"},"agent_actions":{"view_html":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM","download_json":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM.json","view_paper":"https://pith.science/paper/6W27IB7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.04737&json=true","fetch_graph":"https://pith.science/api/pith-number/6W27IB7SZIS2CCHSG4CINIDZLM/graph.json","fetch_events":"https://pith.science/api/pith-number/6W27IB7SZIS2CCHSG4CINIDZLM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM/action/storage_attestation","attest_author":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM/action/author_attestation","sign_citation":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM/action/citation_signature","submit_replication":"https://pith.science/pith/6W27IB7SZIS2CCHSG4CINIDZLM/action/replication_record"}},"created_at":"2026-07-05T08:53:42.899477+00:00","updated_at":"2026-07-05T08:53:42.899477+00:00"}