{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LFKBAIPQMCENWGCAXNNZEONLTX","short_pith_number":"pith:LFKBAIPQ","schema_version":"1.0","canonical_sha256":"59541021f06088db1840bb5b9239ab9df4ba7eeb4e031dce3e3a9413ee967c25","source":{"kind":"arxiv","id":"2407.15835","version":3},"attestation_state":"computed","paper":{"title":"dMel: Speech Tokenization made Simple","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Navdeep Jaitly, Richard He Bai, Ruixiang Zhang, Tatiana Likhomanenko, Zakaria Aldeneh, Zijin Gu","submitted_at":"2024-07-22T17:51:53Z","abstract_excerpt":"Large language models have revolutionized natural language processing by leveraging self-supervised pretraining on vast textual data. Inspired by this success, researchers have investigated various compression-based speech tokenization methods to discretize continuous speech signals, enabling the application of language modeling techniques to discrete tokens. However, audio compressor introduces additional complexity and computational cost, and often fail on out-of-domain audio signals. In this work, we introduce a novel speech representation (dmel) that discretizes mel-filterbank channels int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.15835","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-22T17:51:53Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"e675b4abe8df40197450695d2563f365301e3e620b1202e10ad2680aa564a4f8","abstract_canon_sha256":"0503b25ccfc737b3d67d7308e42522bf50ca0ff70ed2257b74194f236fb373ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:54.235755Z","signature_b64":"6C83IajrmwHVKIQtvzaZSPJ229zgPY7xL/+Br31AxSX9S6BX/zSyoQd+FWjMCn+gnKjcmBtDULLy0nNfPugMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59541021f06088db1840bb5b9239ab9df4ba7eeb4e031dce3e3a9413ee967c25","last_reissued_at":"2026-07-05T11:06:54.235279Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:54.235279Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"dMel: Speech Tokenization made Simple","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Navdeep Jaitly, Richard He Bai, Ruixiang Zhang, Tatiana Likhomanenko, Zakaria Aldeneh, Zijin Gu","submitted_at":"2024-07-22T17:51:53Z","abstract_excerpt":"Large language models have revolutionized natural language processing by leveraging self-supervised pretraining on vast textual data. Inspired by this success, researchers have investigated various compression-based speech tokenization methods to discretize continuous speech signals, enabling the application of language modeling techniques to discrete tokens. However, audio compressor introduces additional complexity and computational cost, and often fail on out-of-domain audio signals. In this work, we introduce a novel speech representation (dmel) that discretizes mel-filterbank channels int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.15835","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.15835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.15835","created_at":"2026-07-05T11:06:54.235344+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.15835v3","created_at":"2026-07-05T11:06:54.235344+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.15835","created_at":"2026-07-05T11:06:54.235344+00:00"},{"alias_kind":"pith_short_12","alias_value":"LFKBAIPQMCEN","created_at":"2026-07-05T11:06:54.235344+00:00"},{"alias_kind":"pith_short_16","alias_value":"LFKBAIPQMCENWGCA","created_at":"2026-07-05T11:06:54.235344+00:00"},{"alias_kind":"pith_short_8","alias_value":"LFKBAIPQ","created_at":"2026-07-05T11:06:54.235344+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10231","citing_title":"LLM can Read Spectrogram: Encoder-free Speech-Language Modeling","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12940","citing_title":"Self-Guidance: Enhancing Neural Codecs via Decoder Manifold Alignment","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29859","citing_title":"MELD: Mel-Spectrogram-Based Speech Language Modeling with Discrete Latent Variables","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2411.17690","citing_title":"Mechanisms of Multimodal Synchronization: Insights from Decoder-Based Video-Text-to-Speech Synthesis","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06885","citing_title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24416","citing_title":"Scaling Properties of Continuous Diffusion Spoken Language Models","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX","json":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX.json","graph_json":"https://pith.science/api/pith-number/LFKBAIPQMCENWGCAXNNZEONLTX/graph.json","events_json":"https://pith.science/api/pith-number/LFKBAIPQMCENWGCAXNNZEONLTX/events.json","paper":"https://pith.science/paper/LFKBAIPQ"},"agent_actions":{"view_html":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX","download_json":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX.json","view_paper":"https://pith.science/paper/LFKBAIPQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.15835&json=true","fetch_graph":"https://pith.science/api/pith-number/LFKBAIPQMCENWGCAXNNZEONLTX/graph.json","fetch_events":"https://pith.science/api/pith-number/LFKBAIPQMCENWGCAXNNZEONLTX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX/action/storage_attestation","attest_author":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX/action/author_attestation","sign_citation":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX/action/citation_signature","submit_replication":"https://pith.science/pith/LFKBAIPQMCENWGCAXNNZEONLTX/action/replication_record"}},"created_at":"2026-07-05T11:06:54.235344+00:00","updated_at":"2026-07-05T11:06:54.235344+00:00"}