{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:7GYEUI3YWPFXZVSZCS5IFQLL7H","short_pith_number":"pith:7GYEUI3Y","schema_version":"1.0","canonical_sha256":"f9b04a2378b3cb7cd65914ba82c16bf9d3e573ae420181b09901588352a0b7e0","source":{"kind":"arxiv","id":"2010.00475","version":2},"attestation_state":"computed","paper":{"title":"FSD50K: An Open Dataset of Human-Labeled Sound Events","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","eess.AS","stat.ML"],"primary_cat":"cs.SD","authors_text":"Eduardo Fonseca, Frederic Font, Jordi Pons, Xavier Favory, Xavier Serra","submitted_at":"2020-10-01T15:07:25Z","abstract_excerpt":"Most existing datasets for sound event recognition (SER) are relatively small and/or domain-specific, with the exception of AudioSet, based on over 2M tracks from YouTube videos and encompassing over 500 sound classes. However, AudioSet is not an open dataset as its official release consists of pre-computed audio features. Downloading the original audio tracks can be problematic due to YouTube videos gradually disappearing and usage rights issues. To provide an alternative benchmark dataset and thus foster SER research, we introduce FSD50K, an open dataset containing over 51k audio clips total"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.00475","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2020-10-01T15:07:25Z","cross_cats_sorted":["cs.LG","eess.AS","stat.ML"],"title_canon_sha256":"5c94b142bfbb18ab9e0d0464b4c0c860a0bc81192868c4bbc15ee7178b112282","abstract_canon_sha256":"d8b92c470d3d7882e0b51f13263b278bb980656c2ed0c3aec85b9803beceea96"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:17:13.063174Z","signature_b64":"t719p1HWxlnQweqjS+o9mpiiZUUyuQzV8MAfjS6kuS6qa5ZkcHw3SSRg8o0WTi9xkKVpG49oxKMXaDPPANOwDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9b04a2378b3cb7cd65914ba82c16bf9d3e573ae420181b09901588352a0b7e0","last_reissued_at":"2026-07-05T04:17:13.062708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:17:13.062708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FSD50K: An Open Dataset of Human-Labeled Sound Events","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","eess.AS","stat.ML"],"primary_cat":"cs.SD","authors_text":"Eduardo Fonseca, Frederic Font, Jordi Pons, Xavier Favory, Xavier Serra","submitted_at":"2020-10-01T15:07:25Z","abstract_excerpt":"Most existing datasets for sound event recognition (SER) are relatively small and/or domain-specific, with the exception of AudioSet, based on over 2M tracks from YouTube videos and encompassing over 500 sound classes. However, AudioSet is not an open dataset as its official release consists of pre-computed audio features. Downloading the original audio tracks can be problematic due to YouTube videos gradually disappearing and usage rights issues. To provide an alternative benchmark dataset and thus foster SER research, we introduce FSD50K, an open dataset containing over 51k audio clips total"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.00475","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.00475/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.00475","created_at":"2026-07-05T04:17:13.062767+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.00475v2","created_at":"2026-07-05T04:17:13.062767+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.00475","created_at":"2026-07-05T04:17:13.062767+00:00"},{"alias_kind":"pith_short_12","alias_value":"7GYEUI3YWPFX","created_at":"2026-07-05T04:17:13.062767+00:00"},{"alias_kind":"pith_short_16","alias_value":"7GYEUI3YWPFXZVSZ","created_at":"2026-07-05T04:17:13.062767+00:00"},{"alias_kind":"pith_short_8","alias_value":"7GYEUI3Y","created_at":"2026-07-05T04:17:13.062767+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10046","citing_title":"Inside the Latent Flow: Causal Deciphering of Attention Dynamics in Audio Separation Foundation Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11717","citing_title":"CodecSep: Prompt-Driven Universal Sound Separation on Neural Audio Codec Latents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02954","citing_title":"The World is Not Mono: Enabling Spatial Understanding in Large Audio-Language Models","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H","json":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H.json","graph_json":"https://pith.science/api/pith-number/7GYEUI3YWPFXZVSZCS5IFQLL7H/graph.json","events_json":"https://pith.science/api/pith-number/7GYEUI3YWPFXZVSZCS5IFQLL7H/events.json","paper":"https://pith.science/paper/7GYEUI3Y"},"agent_actions":{"view_html":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H","download_json":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H.json","view_paper":"https://pith.science/paper/7GYEUI3Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.00475&json=true","fetch_graph":"https://pith.science/api/pith-number/7GYEUI3YWPFXZVSZCS5IFQLL7H/graph.json","fetch_events":"https://pith.science/api/pith-number/7GYEUI3YWPFXZVSZCS5IFQLL7H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H/action/storage_attestation","attest_author":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H/action/author_attestation","sign_citation":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H/action/citation_signature","submit_replication":"https://pith.science/pith/7GYEUI3YWPFXZVSZCS5IFQLL7H/action/replication_record"}},"created_at":"2026-07-05T04:17:13.062767+00:00","updated_at":"2026-07-05T04:17:13.062767+00:00"}