{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:G7OSAG5FZM32X2ABG3R47COJQQ","short_pith_number":"pith:G7OSAG5F","schema_version":"1.0","canonical_sha256":"37dd201ba5cb37abe80136e3cf89c984318d28f74527f76d139c34a2fca4a044","source":{"kind":"arxiv","id":"2107.04954","version":2},"attestation_state":"computed","paper":{"title":"ReconVAT: A Semi-Supervised Automatic Music Transcription Framework for Low-Resource Real-World Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Dorien Herremans, Kin Wai Cheuk, Li Su","submitted_at":"2021-07-11T03:25:58Z","abstract_excerpt":"Most of the current supervised automatic music transcription (AMT) models lack the ability to generalize. This means that they have trouble transcribing real-world music recordings from diverse musical genres that are not presented in the labelled training data. In this paper, we propose a semi-supervised framework, ReconVAT, which solves this issue by leveraging the huge amount of available unlabelled music recordings. The proposed ReconVAT uses reconstruction loss and virtual adversarial training. When combined with existing U-net models for AMT, ReconVAT achieves competitive results on comm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.04954","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2021-07-11T03:25:58Z","cross_cats_sorted":["cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"1840c0c53729bd22a88f8dad7d691c7de6812b88f0ba68f7944b89002b485d81","abstract_canon_sha256":"4be4f947b05350909908cfae2d982ffd9ed32b9b20b6764ddd56ffea71952c2d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:01:37.032271Z","signature_b64":"jW3mBF8MVGW3eynx9u01wjO3tSK6gypdp7gSbhl52M5tJW8M0n7wsnSj9VCg+76oGbLiwVHjXOZgrCkmI1lBBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"37dd201ba5cb37abe80136e3cf89c984318d28f74527f76d139c34a2fca4a044","last_reissued_at":"2026-07-05T03:01:37.031840Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:01:37.031840Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReconVAT: A Semi-Supervised Automatic Music Transcription Framework for Low-Resource Real-World Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Dorien Herremans, Kin Wai Cheuk, Li Su","submitted_at":"2021-07-11T03:25:58Z","abstract_excerpt":"Most of the current supervised automatic music transcription (AMT) models lack the ability to generalize. This means that they have trouble transcribing real-world music recordings from diverse musical genres that are not presented in the labelled training data. In this paper, we propose a semi-supervised framework, ReconVAT, which solves this issue by leveraging the huge amount of available unlabelled music recordings. The proposed ReconVAT uses reconstruction loss and virtual adversarial training. When combined with existing U-net models for AMT, ReconVAT achieves competitive results on comm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.04954","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.04954/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.04954","created_at":"2026-07-05T03:01:37.031897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.04954v2","created_at":"2026-07-05T03:01:37.031897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.04954","created_at":"2026-07-05T03:01:37.031897+00:00"},{"alias_kind":"pith_short_12","alias_value":"G7OSAG5FZM32","created_at":"2026-07-05T03:01:37.031897+00:00"},{"alias_kind":"pith_short_16","alias_value":"G7OSAG5FZM32X2AB","created_at":"2026-07-05T03:01:37.031897+00:00"},{"alias_kind":"pith_short_8","alias_value":"G7OSAG5F","created_at":"2026-07-05T03:01:37.031897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24193","citing_title":"Music Transcription with (Almost) No Supervision","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ","json":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ.json","graph_json":"https://pith.science/api/pith-number/G7OSAG5FZM32X2ABG3R47COJQQ/graph.json","events_json":"https://pith.science/api/pith-number/G7OSAG5FZM32X2ABG3R47COJQQ/events.json","paper":"https://pith.science/paper/G7OSAG5F"},"agent_actions":{"view_html":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ","download_json":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ.json","view_paper":"https://pith.science/paper/G7OSAG5F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.04954&json=true","fetch_graph":"https://pith.science/api/pith-number/G7OSAG5FZM32X2ABG3R47COJQQ/graph.json","fetch_events":"https://pith.science/api/pith-number/G7OSAG5FZM32X2ABG3R47COJQQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ/action/storage_attestation","attest_author":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ/action/author_attestation","sign_citation":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ/action/citation_signature","submit_replication":"https://pith.science/pith/G7OSAG5FZM32X2ABG3R47COJQQ/action/replication_record"}},"created_at":"2026-07-05T03:01:37.031897+00:00","updated_at":"2026-07-05T03:01:37.031897+00:00"}