{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3ZTS2OPVBE6K3VRKVN3HWPC4QO","short_pith_number":"pith:3ZTS2OPV","schema_version":"1.0","canonical_sha256":"de672d39f5093cadd62aab767b3c5c83b2e19f1e05b8f71d5ddede3f6a969d71","source":{"kind":"arxiv","id":"2502.15602","version":2},"attestation_state":"computed","paper":{"title":"KAD: No More FAD! An Effective and Efficient Evaluation Metric for Audio Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ben Sangbae Chon, Juhan Nam, Junwon Lee, Keunwoo Choi, Pilsun Eu, Yoonjin Chung","submitted_at":"2025-02-21T17:19:15Z","abstract_excerpt":"Although being widely adopted for evaluating generated audio signals, the Fr\\'echet Audio Distance (FAD) suffers from significant limitations, including reliance on Gaussian assumptions, sensitivity to sample size, and high computational complexity. As an alternative, we introduce the Kernel Audio Distance (KAD), a novel, distribution-free, unbiased, and computationally efficient metric based on Maximum Mean Discrepancy (MMD). Through analysis and empirical validation, we demonstrate KAD's advantages: (1) faster convergence with smaller sample sizes, enabling reliable evaluation with limited d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.15602","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2025-02-21T17:19:15Z","cross_cats_sorted":["cs.AI","cs.LG","eess.AS"],"title_canon_sha256":"f01c97bcef2decc9427845c7f623786c6add7f7da4d382498baba2535905e7fa","abstract_canon_sha256":"ea5b68c5aa8a18b37c223c95282950fd9e773ba88bad06af6f3a89d0e8d71fdc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:27:25.898535Z","signature_b64":"vc2BgFoSsjesBHNtrDB8LTSj72f+ynOVcyi9l57jDElAMw5gGYR0PQdbkbkhwOYTzsnjDGpGiIlAnPTn/UvzCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de672d39f5093cadd62aab767b3c5c83b2e19f1e05b8f71d5ddede3f6a969d71","last_reissued_at":"2026-07-05T10:27:25.898035Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:27:25.898035Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KAD: No More FAD! An Effective and Efficient Evaluation Metric for Audio Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ben Sangbae Chon, Juhan Nam, Junwon Lee, Keunwoo Choi, Pilsun Eu, Yoonjin Chung","submitted_at":"2025-02-21T17:19:15Z","abstract_excerpt":"Although being widely adopted for evaluating generated audio signals, the Fr\\'echet Audio Distance (FAD) suffers from significant limitations, including reliance on Gaussian assumptions, sensitivity to sample size, and high computational complexity. As an alternative, we introduce the Kernel Audio Distance (KAD), a novel, distribution-free, unbiased, and computationally efficient metric based on Maximum Mean Discrepancy (MMD). Through analysis and empirical validation, we demonstrate KAD's advantages: (1) faster convergence with smaller sample sizes, enabling reliable evaluation with limited d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15602","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.15602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.15602","created_at":"2026-07-05T10:27:25.898095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.15602v2","created_at":"2026-07-05T10:27:25.898095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15602","created_at":"2026-07-05T10:27:25.898095+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ZTS2OPVBE6K","created_at":"2026-07-05T10:27:25.898095+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ZTS2OPVBE6K3VRK","created_at":"2026-07-05T10:27:25.898095+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ZTS2OPV","created_at":"2026-07-05T10:27:25.898095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06027","citing_title":"Fr\\'echet Distance Loss on Speech Representations for Text-to-Speech Synthesis","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12282","citing_title":"PianoKontext: Expressive Performance Rendering from Deadpan Context","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11581","citing_title":"Sensitivity Analysis of Generative Spatial Audio Metrics: A Study on Responsiveness, Smoothness, and Symmetry","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10203","citing_title":"Polyphonia: Zero-Shot Timbre Transfer in Polyphonic Music with Acoustic-Informed Attention Calibration","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05554","citing_title":"Optimal Transport Audio Distance with Learned Riemannian Ground Metrics","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO","json":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO.json","graph_json":"https://pith.science/api/pith-number/3ZTS2OPVBE6K3VRKVN3HWPC4QO/graph.json","events_json":"https://pith.science/api/pith-number/3ZTS2OPVBE6K3VRKVN3HWPC4QO/events.json","paper":"https://pith.science/paper/3ZTS2OPV"},"agent_actions":{"view_html":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO","download_json":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO.json","view_paper":"https://pith.science/paper/3ZTS2OPV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.15602&json=true","fetch_graph":"https://pith.science/api/pith-number/3ZTS2OPVBE6K3VRKVN3HWPC4QO/graph.json","fetch_events":"https://pith.science/api/pith-number/3ZTS2OPVBE6K3VRKVN3HWPC4QO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO/action/storage_attestation","attest_author":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO/action/author_attestation","sign_citation":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO/action/citation_signature","submit_replication":"https://pith.science/pith/3ZTS2OPVBE6K3VRKVN3HWPC4QO/action/replication_record"}},"created_at":"2026-07-05T10:27:25.898095+00:00","updated_at":"2026-07-05T10:27:25.898095+00:00"}