{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KXUVLO2LRJLXYAZDWQDZ7MNEIY","short_pith_number":"pith:KXUVLO2L","schema_version":"1.0","canonical_sha256":"55e955bb4b8a577c0323b4079fb1a446138603030cd2fceca271ecb199945789","source":{"kind":"arxiv","id":"2404.17161","version":1},"attestation_state":"computed","paper":{"title":"An Investigation of Time-Frequency Representation Discriminators for High-Fidelity Vocoder","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS","eess.SP"],"primary_cat":"cs.SD","authors_text":"Haizhou Li, Liumeng Xue, Xueyao Zhang, Yicheng Gu, Zhizheng Wu","submitted_at":"2024-04-26T05:11:03Z","abstract_excerpt":"Generative Adversarial Network (GAN) based vocoders are superior in both inference speed and synthesis quality when reconstructing an audible waveform from an acoustic representation. This study focuses on improving the discriminator for GAN-based vocoders. Most existing Time-Frequency Representation (TFR)-based discriminators are rooted in Short-Time Fourier Transform (STFT), which owns a constant Time-Frequency (TF) resolution, linearly scaled center frequencies, and a fixed decomposition basis, making it incompatible with signals like singing voices that require dynamic attention for differ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.17161","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-04-26T05:11:03Z","cross_cats_sorted":["eess.AS","eess.SP"],"title_canon_sha256":"25c380aecc9ba17b66ebd447a3fff576169f2659a9be508808cdc6f912b6efee","abstract_canon_sha256":"6aaef44cb45ffd06ef0d48d2ed2d5f410fc81ae5d6263028d722d70096bf7937"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:12:33.725179Z","signature_b64":"II+d6nPkxNJCCi0qNtTTsG7IE0u4Ab6zd/swVNE5ia4llvuy7V/PGm8PofsCavuImVDvBbHOy7gGCnP+06jeAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"55e955bb4b8a577c0323b4079fb1a446138603030cd2fceca271ecb199945789","last_reissued_at":"2026-07-05T08:12:33.724762Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:12:33.724762Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Investigation of Time-Frequency Representation Discriminators for High-Fidelity Vocoder","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS","eess.SP"],"primary_cat":"cs.SD","authors_text":"Haizhou Li, Liumeng Xue, Xueyao Zhang, Yicheng Gu, Zhizheng Wu","submitted_at":"2024-04-26T05:11:03Z","abstract_excerpt":"Generative Adversarial Network (GAN) based vocoders are superior in both inference speed and synthesis quality when reconstructing an audible waveform from an acoustic representation. This study focuses on improving the discriminator for GAN-based vocoders. Most existing Time-Frequency Representation (TFR)-based discriminators are rooted in Short-Time Fourier Transform (STFT), which owns a constant Time-Frequency (TF) resolution, linearly scaled center frequencies, and a fixed decomposition basis, making it incompatible with signals like singing voices that require dynamic attention for differ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.17161","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.17161/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.17161","created_at":"2026-07-05T08:12:33.724818+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.17161v1","created_at":"2026-07-05T08:12:33.724818+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.17161","created_at":"2026-07-05T08:12:33.724818+00:00"},{"alias_kind":"pith_short_12","alias_value":"KXUVLO2LRJLX","created_at":"2026-07-05T08:12:33.724818+00:00"},{"alias_kind":"pith_short_16","alias_value":"KXUVLO2LRJLXYAZD","created_at":"2026-07-05T08:12:33.724818+00:00"},{"alias_kind":"pith_short_8","alias_value":"KXUVLO2L","created_at":"2026-07-05T08:12:33.724818+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.09325","citing_title":"SingNet: Towards a Large-Scale, Diverse, and In-the-Wild Singing Voice Dataset","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY","json":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY.json","graph_json":"https://pith.science/api/pith-number/KXUVLO2LRJLXYAZDWQDZ7MNEIY/graph.json","events_json":"https://pith.science/api/pith-number/KXUVLO2LRJLXYAZDWQDZ7MNEIY/events.json","paper":"https://pith.science/paper/KXUVLO2L"},"agent_actions":{"view_html":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY","download_json":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY.json","view_paper":"https://pith.science/paper/KXUVLO2L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.17161&json=true","fetch_graph":"https://pith.science/api/pith-number/KXUVLO2LRJLXYAZDWQDZ7MNEIY/graph.json","fetch_events":"https://pith.science/api/pith-number/KXUVLO2LRJLXYAZDWQDZ7MNEIY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY/action/storage_attestation","attest_author":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY/action/author_attestation","sign_citation":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY/action/citation_signature","submit_replication":"https://pith.science/pith/KXUVLO2LRJLXYAZDWQDZ7MNEIY/action/replication_record"}},"created_at":"2026-07-05T08:12:33.724818+00:00","updated_at":"2026-07-05T08:12:33.724818+00:00"}