{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:QWBEHWIRNKMP7DULUDD24U7W3T","short_pith_number":"pith:QWBEHWIR","schema_version":"1.0","canonical_sha256":"858243d9116a98ff8e8ba0c7ae53f6dceafecec61cee15da9d939c3244e51547","source":{"kind":"arxiv","id":"2210.03799","version":1},"attestation_state":"computed","paper":{"title":"Supervised and Unsupervised Learning of Audio Representations for Music Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Andreas F. Ehmann, Fabien Gouyon, Filip Korzeniowski, Matthew C. McCallum, Sergio Oramas","submitted_at":"2022-10-07T20:07:35Z","abstract_excerpt":"In this work, we provide a broad comparative analysis of strategies for pre-training audio understanding models for several tasks in the music domain, including labelling of genre, era, origin, mood, instrumentation, key, pitch, vocal characteristics, tempo and sonority. Specifically, we explore how the domain of pre-training datasets (music or generic audio) and the pre-training methodology (supervised or unsupervised) affects the adequacy of the resulting audio embeddings for downstream tasks.\n  We show that models trained via supervised learning on large-scale expert-annotated music dataset"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.03799","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2022-10-07T20:07:35Z","cross_cats_sorted":["cs.AI","cs.IR","cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"2977cd72083323a3dfdb3c4d7b1ecaddca6ba15b71ad7ae0ce558e985a2e58fc","abstract_canon_sha256":"43b387a8b0e2c1c32b3fcadc341a13ee20cd703b0a11fc8802e72de9b7ce0942"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:32.357421Z","signature_b64":"FWaN5Z3BUBWIMiMr/LCKCpLyCbIZYv5JyQbhxwIwca25D+Qqcyi/Rmi5oplirqtvCXmYQMNNC3JQfd3LcapuAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"858243d9116a98ff8e8ba0c7ae53f6dceafecec61cee15da9d939c3244e51547","last_reissued_at":"2026-07-05T05:04:32.356961Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:32.356961Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Supervised and Unsupervised Learning of Audio Representations for Music Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Andreas F. Ehmann, Fabien Gouyon, Filip Korzeniowski, Matthew C. McCallum, Sergio Oramas","submitted_at":"2022-10-07T20:07:35Z","abstract_excerpt":"In this work, we provide a broad comparative analysis of strategies for pre-training audio understanding models for several tasks in the music domain, including labelling of genre, era, origin, mood, instrumentation, key, pitch, vocal characteristics, tempo and sonority. Specifically, we explore how the domain of pre-training datasets (music or generic audio) and the pre-training methodology (supervised or unsupervised) affects the adequacy of the resulting audio embeddings for downstream tasks.\n  We show that models trained via supervised learning on large-scale expert-annotated music dataset"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.03799","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.03799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.03799","created_at":"2026-07-05T05:04:32.357019+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.03799v1","created_at":"2026-07-05T05:04:32.357019+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.03799","created_at":"2026-07-05T05:04:32.357019+00:00"},{"alias_kind":"pith_short_12","alias_value":"QWBEHWIRNKMP","created_at":"2026-07-05T05:04:32.357019+00:00"},{"alias_kind":"pith_short_16","alias_value":"QWBEHWIRNKMP7DUL","created_at":"2026-07-05T05:04:32.357019+00:00"},{"alias_kind":"pith_short_8","alias_value":"QWBEHWIR","created_at":"2026-07-05T05:04:32.357019+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00777","citing_title":"Evaluating Pretrained Music Embeddings for Cross-Performance Jazz Standard Recognition","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23077","citing_title":"Adopting State-of-the-Art Pretrained Audio Representations for Music Recommender Systems","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T","json":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T.json","graph_json":"https://pith.science/api/pith-number/QWBEHWIRNKMP7DULUDD24U7W3T/graph.json","events_json":"https://pith.science/api/pith-number/QWBEHWIRNKMP7DULUDD24U7W3T/events.json","paper":"https://pith.science/paper/QWBEHWIR"},"agent_actions":{"view_html":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T","download_json":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T.json","view_paper":"https://pith.science/paper/QWBEHWIR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.03799&json=true","fetch_graph":"https://pith.science/api/pith-number/QWBEHWIRNKMP7DULUDD24U7W3T/graph.json","fetch_events":"https://pith.science/api/pith-number/QWBEHWIRNKMP7DULUDD24U7W3T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T/action/storage_attestation","attest_author":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T/action/author_attestation","sign_citation":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T/action/citation_signature","submit_replication":"https://pith.science/pith/QWBEHWIRNKMP7DULUDD24U7W3T/action/replication_record"}},"created_at":"2026-07-05T05:04:32.357019+00:00","updated_at":"2026-07-05T05:04:32.357019+00:00"}