{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HDEJYGSRNADTYG6C4KHF6L3VTV","short_pith_number":"pith:HDEJYGSR","schema_version":"1.0","canonical_sha256":"38c89c1a5168073c1bc2e28e5f2f759d4765f6f20436e626326cba1ca9b20af1","source":{"kind":"arxiv","id":"2309.10922","version":1},"attestation_state":"computed","paper":{"title":"Discrete Audio Representation as an Alternative to Mel-Spectrograms for Speaker and Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Boris Ginsburg, Jagadeesh Balam, Krishna C. Puvvada, Kunal Dhawan, Nithin Rao Koluguri","submitted_at":"2023-09-19T20:49:05Z","abstract_excerpt":"Discrete audio representation, aka audio tokenization, has seen renewed interest driven by its potential to facilitate the application of text language modeling approaches in audio domain. To this end, various compression and representation-learning based tokenization schemes have been proposed. However, there is limited investigation into the performance of compression-based audio tokens compared to well-established mel-spectrogram features across various speaker and speech related tasks. In this paper, we evaluate compression based audio tokens on three tasks: Speaker Verification, Diarizati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.10922","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2023-09-19T20:49:05Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"cd0e3803ab82148183ae1b0cad44637f9af8aef584142376091101e6d27afc1a","abstract_canon_sha256":"34fe5a074e74a10454b154b232cb9bde99462ade8d3792ac860e02d6991814f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:52:28.577551Z","signature_b64":"Lfh/vGgTMP8Zar6+7eoxr8cyIQ6khMlJDU6sNKddee55YAB8XRYL4a6Jh9nnF6P2r3l5IGHj0A5vji4NtojRCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38c89c1a5168073c1bc2e28e5f2f759d4765f6f20436e626326cba1ca9b20af1","last_reissued_at":"2026-07-05T06:52:28.577058Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:52:28.577058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Discrete Audio Representation as an Alternative to Mel-Spectrograms for Speaker and Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Boris Ginsburg, Jagadeesh Balam, Krishna C. Puvvada, Kunal Dhawan, Nithin Rao Koluguri","submitted_at":"2023-09-19T20:49:05Z","abstract_excerpt":"Discrete audio representation, aka audio tokenization, has seen renewed interest driven by its potential to facilitate the application of text language modeling approaches in audio domain. To this end, various compression and representation-learning based tokenization schemes have been proposed. However, there is limited investigation into the performance of compression-based audio tokens compared to well-established mel-spectrogram features across various speaker and speech related tasks. In this paper, we evaluate compression based audio tokens on three tasks: Speaker Verification, Diarizati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.10922","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.10922/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.10922","created_at":"2026-07-05T06:52:28.577113+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.10922v1","created_at":"2026-07-05T06:52:28.577113+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.10922","created_at":"2026-07-05T06:52:28.577113+00:00"},{"alias_kind":"pith_short_12","alias_value":"HDEJYGSRNADT","created_at":"2026-07-05T06:52:28.577113+00:00"},{"alias_kind":"pith_short_16","alias_value":"HDEJYGSRNADTYG6C","created_at":"2026-07-05T06:52:28.577113+00:00"},{"alias_kind":"pith_short_8","alias_value":"HDEJYGSR","created_at":"2026-07-05T06:52:28.577113+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.02771","citing_title":"Analysis of Speaker Verification Performance Trade-offs with Neural Audio Codec Transmission","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV","json":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV.json","graph_json":"https://pith.science/api/pith-number/HDEJYGSRNADTYG6C4KHF6L3VTV/graph.json","events_json":"https://pith.science/api/pith-number/HDEJYGSRNADTYG6C4KHF6L3VTV/events.json","paper":"https://pith.science/paper/HDEJYGSR"},"agent_actions":{"view_html":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV","download_json":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV.json","view_paper":"https://pith.science/paper/HDEJYGSR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.10922&json=true","fetch_graph":"https://pith.science/api/pith-number/HDEJYGSRNADTYG6C4KHF6L3VTV/graph.json","fetch_events":"https://pith.science/api/pith-number/HDEJYGSRNADTYG6C4KHF6L3VTV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV/action/storage_attestation","attest_author":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV/action/author_attestation","sign_citation":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV/action/citation_signature","submit_replication":"https://pith.science/pith/HDEJYGSRNADTYG6C4KHF6L3VTV/action/replication_record"}},"created_at":"2026-07-05T06:52:28.577113+00:00","updated_at":"2026-07-05T06:52:28.577113+00:00"}