{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:44F2GNATNTYZHJN5J45EZPTM6N","short_pith_number":"pith:44F2GNAT","schema_version":"1.0","canonical_sha256":"e70ba334136cf193a5bd4f3a4cbe6cf341b29e92c8b238ff66d6468dfd943f30","source":{"kind":"arxiv","id":"2101.08596","version":1},"attestation_state":"computed","paper":{"title":"LEAF: A Learnable Frontend for Audio Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"F\\'elix de Chaumont Quitry, Marco Tagliasacchi, Neil Zeghidour, Olivier Teboul","submitted_at":"2021-01-21T13:25:58Z","abstract_excerpt":"Mel-filterbanks are fixed, engineered audio features which emulate human perception and have been used through the history of audio understanding up to today. However, their undeniable qualities are counterbalanced by the fundamental limitations of handmade representations. In this work we show that we can train a single learnable frontend that outperforms mel-filterbanks on a wide range of audio signals, including speech, music, audio events and animal sounds, providing a general-purpose learned frontend for audio classification. To do so, we introduce a new principled, lightweight, fully lea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.08596","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2021-01-21T13:25:58Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"eafff9c7a95d9b6348fa04d453f5af00b6e3c78b703d76867217943756baa060","abstract_canon_sha256":"09f4f3bc73bc07cf321330f922c39f068023efc2d957bc23fb86e5bcf2610a09"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:08:37.005950Z","signature_b64":"n/MUXqONdbH+rE2Y/eYl1lUQMF5o8BVCHhuPOqkzWmlfxikhs4afcm4cWN3P4kpQ7minR8R2WSuxK51Cn0/YDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e70ba334136cf193a5bd4f3a4cbe6cf341b29e92c8b238ff66d6468dfd943f30","last_reissued_at":"2026-07-05T02:08:37.005411Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:08:37.005411Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LEAF: A Learnable Frontend for Audio Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"F\\'elix de Chaumont Quitry, Marco Tagliasacchi, Neil Zeghidour, Olivier Teboul","submitted_at":"2021-01-21T13:25:58Z","abstract_excerpt":"Mel-filterbanks are fixed, engineered audio features which emulate human perception and have been used through the history of audio understanding up to today. However, their undeniable qualities are counterbalanced by the fundamental limitations of handmade representations. In this work we show that we can train a single learnable frontend that outperforms mel-filterbanks on a wide range of audio signals, including speech, music, audio events and animal sounds, providing a general-purpose learned frontend for audio classification. To do so, we introduce a new principled, lightweight, fully lea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.08596","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.08596/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.08596","created_at":"2026-07-05T02:08:37.005466+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.08596v1","created_at":"2026-07-05T02:08:37.005466+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.08596","created_at":"2026-07-05T02:08:37.005466+00:00"},{"alias_kind":"pith_short_12","alias_value":"44F2GNATNTYZ","created_at":"2026-07-05T02:08:37.005466+00:00"},{"alias_kind":"pith_short_16","alias_value":"44F2GNATNTYZHJN5","created_at":"2026-07-05T02:08:37.005466+00:00"},{"alias_kind":"pith_short_8","alias_value":"44F2GNAT","created_at":"2026-07-05T02:08:37.005466+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08545","citing_title":"Structural Bottlenecks on Frequency Representation in End-to-End Audio Models","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.29518","citing_title":"Harvesting AI Computation at the Edge via Generic Approximation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19209","citing_title":"Audio Spoof Detection with GaborNet","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10503","citing_title":"Cross-Cultural Bias in Mel-Scale Representations: Evidence and Alternatives from Speech and Music","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N","json":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N.json","graph_json":"https://pith.science/api/pith-number/44F2GNATNTYZHJN5J45EZPTM6N/graph.json","events_json":"https://pith.science/api/pith-number/44F2GNATNTYZHJN5J45EZPTM6N/events.json","paper":"https://pith.science/paper/44F2GNAT"},"agent_actions":{"view_html":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N","download_json":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N.json","view_paper":"https://pith.science/paper/44F2GNAT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.08596&json=true","fetch_graph":"https://pith.science/api/pith-number/44F2GNATNTYZHJN5J45EZPTM6N/graph.json","fetch_events":"https://pith.science/api/pith-number/44F2GNATNTYZHJN5J45EZPTM6N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N/action/storage_attestation","attest_author":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N/action/author_attestation","sign_citation":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N/action/citation_signature","submit_replication":"https://pith.science/pith/44F2GNATNTYZHJN5J45EZPTM6N/action/replication_record"}},"created_at":"2026-07-05T02:08:37.005466+00:00","updated_at":"2026-07-05T02:08:37.005466+00:00"}