{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:AVS2OGZ5IJLC5ATDK6BLAAQMGK","short_pith_number":"pith:AVS2OGZ5","schema_version":"1.0","canonical_sha256":"0565a71b3d42562e82635782b0020c32a1cf3ecb28ae380835213030d288e5ef","source":{"kind":"arxiv","id":"2109.09227","version":2},"attestation_state":"computed","paper":{"title":"ARCA23K: An audio dataset for investigating open-set label noise","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Andrew Bailey, Mark D. Plumbley, Turab Iqbal, Wenwu Wang, Yin Cao","submitted_at":"2021-09-19T21:10:25Z","abstract_excerpt":"The availability of audio data on sound sharing platforms such as Freesound gives users access to large amounts of annotated audio. Utilising such data for training is becoming increasingly popular, but the problem of label noise that is often prevalent in such datasets requires further investigation. This paper introduces ARCA23K, an Automatically Retrieved and Curated Audio dataset comprised of over 23000 labelled Freesound clips. Unlike past datasets such as FSDKaggle2018 and FSDnoisy18K, ARCA23K facilitates the study of label noise in a more controlled manner. We describe the entire proces"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.09227","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.SD","submitted_at":"2021-09-19T21:10:25Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"0c760f9cb32df2fc22f42f186c41d8d612f84be69130a4a230eb0cbb0e3d6048","abstract_canon_sha256":"ebf308b4bd798caa0cb31a8aea4054da74cd1d39885d5f9417ee04f1ff46417e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:00:18.182395Z","signature_b64":"T4Pq0EYJpfK+KJmn3yrBCdmYKHA4UoaQ6e/Kn0eXLhwTHtT+OHxVjvtJ0tnNJO8iT9EfTdSSVFS8EmE11zvrDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0565a71b3d42562e82635782b0020c32a1cf3ecb28ae380835213030d288e5ef","last_reissued_at":"2026-07-05T04:00:18.181909Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:00:18.181909Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ARCA23K: An audio dataset for investigating open-set label noise","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Andrew Bailey, Mark D. Plumbley, Turab Iqbal, Wenwu Wang, Yin Cao","submitted_at":"2021-09-19T21:10:25Z","abstract_excerpt":"The availability of audio data on sound sharing platforms such as Freesound gives users access to large amounts of annotated audio. Utilising such data for training is becoming increasingly popular, but the problem of label noise that is often prevalent in such datasets requires further investigation. This paper introduces ARCA23K, an Automatically Retrieved and Curated Audio dataset comprised of over 23000 labelled Freesound clips. Unlike past datasets such as FSDKaggle2018 and FSDnoisy18K, ARCA23K facilitates the study of label noise in a more controlled manner. We describe the entire proces"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.09227","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.09227/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.09227","created_at":"2026-07-05T04:00:18.181965+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.09227v2","created_at":"2026-07-05T04:00:18.181965+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.09227","created_at":"2026-07-05T04:00:18.181965+00:00"},{"alias_kind":"pith_short_12","alias_value":"AVS2OGZ5IJLC","created_at":"2026-07-05T04:00:18.181965+00:00"},{"alias_kind":"pith_short_16","alias_value":"AVS2OGZ5IJLC5ATD","created_at":"2026-07-05T04:00:18.181965+00:00"},{"alias_kind":"pith_short_8","alias_value":"AVS2OGZ5","created_at":"2026-07-05T04:00:18.181965+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK","json":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK.json","graph_json":"https://pith.science/api/pith-number/AVS2OGZ5IJLC5ATDK6BLAAQMGK/graph.json","events_json":"https://pith.science/api/pith-number/AVS2OGZ5IJLC5ATDK6BLAAQMGK/events.json","paper":"https://pith.science/paper/AVS2OGZ5"},"agent_actions":{"view_html":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK","download_json":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK.json","view_paper":"https://pith.science/paper/AVS2OGZ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.09227&json=true","fetch_graph":"https://pith.science/api/pith-number/AVS2OGZ5IJLC5ATDK6BLAAQMGK/graph.json","fetch_events":"https://pith.science/api/pith-number/AVS2OGZ5IJLC5ATDK6BLAAQMGK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK/action/storage_attestation","attest_author":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK/action/author_attestation","sign_citation":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK/action/citation_signature","submit_replication":"https://pith.science/pith/AVS2OGZ5IJLC5ATDK6BLAAQMGK/action/replication_record"}},"created_at":"2026-07-05T04:00:18.181965+00:00","updated_at":"2026-07-05T04:00:18.181965+00:00"}