{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:UGFILF3V5Q5AU2GZ562Z5IDSRG","short_pith_number":"pith:UGFILF3V","schema_version":"1.0","canonical_sha256":"a18a859775ec3a0a68d9efb59ea07289a23561217637410991cbc05421c74818","source":{"kind":"arxiv","id":"2607.07579","version":1},"attestation_state":"computed","paper":{"title":"Text-Independent Speaker Verification Using Discrete Audio Tokens","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Junjie Li, Kong Aik Lee, Zheng Liang","submitted_at":"2026-07-08T16:03:44Z","abstract_excerpt":"Neural audio codecs (NACs) enable efficient audio compression and have achieved success in downstream tasks such as speech synthesis. However, their discrete representations consistently underperform traditional spectral features in automatic speaker verification (ASV). We empirically demonstrate that speaker cues are implicitly preserved in discrete tokens but remain underutilized by conventional ASV training paradigms. To address this, we propose a Cross-Feature Knowledge Distillation (CFKD) framework. By guiding the codec-based student to mimic the embedding space of a strong Fbank-based te"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.07579","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2026-07-08T16:03:44Z","cross_cats_sorted":[],"title_canon_sha256":"5b9ff6ff44f327166f6c5bf0f5ad5c83d02cae14ef9220d0e566d5c3ca80297f","abstract_canon_sha256":"8bd766a357a0eb7cc04837dae0b5d6ece6cb3ac0700201d4b216e37ee207a3e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-09T01:20:30.089096Z","signature_b64":"7+asS5JEqdoXilz2jZpNn6POiXxozuwTAl2ma7xQ6Yetki6XHrV2s++LbX2Z/o8buUssXE2yJIB1qDlzB06KCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a18a859775ec3a0a68d9efb59ea07289a23561217637410991cbc05421c74818","last_reissued_at":"2026-07-09T01:20:30.088673Z","signature_status":"signed_v1","first_computed_at":"2026-07-09T01:20:30.088673Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Text-Independent Speaker Verification Using Discrete Audio Tokens","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Junjie Li, Kong Aik Lee, Zheng Liang","submitted_at":"2026-07-08T16:03:44Z","abstract_excerpt":"Neural audio codecs (NACs) enable efficient audio compression and have achieved success in downstream tasks such as speech synthesis. However, their discrete representations consistently underperform traditional spectral features in automatic speaker verification (ASV). We empirically demonstrate that speaker cues are implicitly preserved in discrete tokens but remain underutilized by conventional ASV training paradigms. To address this, we propose a Cross-Feature Knowledge Distillation (CFKD) framework. By guiding the codec-based student to mimic the embedding space of a strong Fbank-based te"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.07579","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.07579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.07579","created_at":"2026-07-09T01:20:30.088745+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.07579v1","created_at":"2026-07-09T01:20:30.088745+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.07579","created_at":"2026-07-09T01:20:30.088745+00:00"},{"alias_kind":"pith_short_12","alias_value":"UGFILF3V5Q5A","created_at":"2026-07-09T01:20:30.088745+00:00"},{"alias_kind":"pith_short_16","alias_value":"UGFILF3V5Q5AU2GZ","created_at":"2026-07-09T01:20:30.088745+00:00"},{"alias_kind":"pith_short_8","alias_value":"UGFILF3V","created_at":"2026-07-09T01:20:30.088745+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07579","citing_title":"Text-Independent Speaker Verification Using Discrete Audio Tokens","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG","json":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG.json","graph_json":"https://pith.science/api/pith-number/UGFILF3V5Q5AU2GZ562Z5IDSRG/graph.json","events_json":"https://pith.science/api/pith-number/UGFILF3V5Q5AU2GZ562Z5IDSRG/events.json","paper":"https://pith.science/paper/UGFILF3V"},"agent_actions":{"view_html":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG","download_json":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG.json","view_paper":"https://pith.science/paper/UGFILF3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.07579&json=true","fetch_graph":"https://pith.science/api/pith-number/UGFILF3V5Q5AU2GZ562Z5IDSRG/graph.json","fetch_events":"https://pith.science/api/pith-number/UGFILF3V5Q5AU2GZ562Z5IDSRG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG/action/storage_attestation","attest_author":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG/action/author_attestation","sign_citation":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG/action/citation_signature","submit_replication":"https://pith.science/pith/UGFILF3V5Q5AU2GZ562Z5IDSRG/action/replication_record"}},"created_at":"2026-07-09T01:20:30.088745+00:00","updated_at":"2026-07-09T01:20:30.088745+00:00"}