{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:75C7JM4LZSNGDUADPRKIFG6BIL","short_pith_number":"pith:75C7JM4L","schema_version":"1.0","canonical_sha256":"ff45f4b38bcc9a61d0037c54829bc142f3d0e2974cea60934a47aaf787df97ea","source":{"kind":"arxiv","id":"2508.05835","version":1},"attestation_state":"computed","paper":{"title":"NanoCodec: Towards High-Quality Ultra Fast Speech LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Ante Juki\\'c, Boris Ginsburg, Edresson Casanova, Jason Li, Paarth Neekhara, Ryan Langman, Shehzeen Hussain, Subhankar Ghosh, Xuesong Yang","submitted_at":"2025-08-07T20:20:32Z","abstract_excerpt":"Large Language Models (LLMs) have significantly advanced audio processing by leveraging audio codecs to discretize audio into tokens, enabling the application of language modeling techniques to speech data. However, existing audio codecs often operate at high frame rates, leading to slow training and inference, particularly for autoregressive models. To address this, there is growing interest in low frame-rate audio codecs, which reduce the number of autoregressive steps required to generate one second of audio. In this paper, we conduct ablation studies to examine the impact of frame rate, bi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.05835","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2025-08-07T20:20:32Z","cross_cats_sorted":["cs.CL","cs.SD"],"title_canon_sha256":"133e0278260452f1ea6dbb64e174657cffc1330f21e9237d54a19a314fadf154","abstract_canon_sha256":"ce1da4c665b97824687a72fef700fecc6b0c52aa34403c59016722123e2de327"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:44.090229Z","signature_b64":"HeWcySLMA6twKP6u9RMMAGYWzZCuyEq6oiTG71XpXmoTE7JgdatM6BheV0rygdZ56eOVELDMC3pXbHZdfkA6CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff45f4b38bcc9a61d0037c54829bc142f3d0e2974cea60934a47aaf787df97ea","last_reissued_at":"2026-07-05T11:50:44.089645Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:44.089645Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NanoCodec: Towards High-Quality Ultra Fast Speech LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Ante Juki\\'c, Boris Ginsburg, Edresson Casanova, Jason Li, Paarth Neekhara, Ryan Langman, Shehzeen Hussain, Subhankar Ghosh, Xuesong Yang","submitted_at":"2025-08-07T20:20:32Z","abstract_excerpt":"Large Language Models (LLMs) have significantly advanced audio processing by leveraging audio codecs to discretize audio into tokens, enabling the application of language modeling techniques to speech data. However, existing audio codecs often operate at high frame rates, leading to slow training and inference, particularly for autoregressive models. To address this, there is growing interest in low frame-rate audio codecs, which reduce the number of autoregressive steps required to generate one second of audio. In this paper, we conduct ablation studies to examine the impact of frame rate, bi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.05835","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.05835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.05835","created_at":"2026-07-05T11:50:44.089704+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.05835v1","created_at":"2026-07-05T11:50:44.089704+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.05835","created_at":"2026-07-05T11:50:44.089704+00:00"},{"alias_kind":"pith_short_12","alias_value":"75C7JM4LZSNG","created_at":"2026-07-05T11:50:44.089704+00:00"},{"alias_kind":"pith_short_16","alias_value":"75C7JM4LZSNGDUAD","created_at":"2026-07-05T11:50:44.089704+00:00"},{"alias_kind":"pith_short_8","alias_value":"75C7JM4L","created_at":"2026-07-05T11:50:44.089704+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06559","citing_title":"IRAF: Interference-Resilient Adaptive Fusion for Noise-Robust End-to-End Full-Duplex Spoken Dialogue Systems","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL","json":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL.json","graph_json":"https://pith.science/api/pith-number/75C7JM4LZSNGDUADPRKIFG6BIL/graph.json","events_json":"https://pith.science/api/pith-number/75C7JM4LZSNGDUADPRKIFG6BIL/events.json","paper":"https://pith.science/paper/75C7JM4L"},"agent_actions":{"view_html":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL","download_json":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL.json","view_paper":"https://pith.science/paper/75C7JM4L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.05835&json=true","fetch_graph":"https://pith.science/api/pith-number/75C7JM4LZSNGDUADPRKIFG6BIL/graph.json","fetch_events":"https://pith.science/api/pith-number/75C7JM4LZSNGDUADPRKIFG6BIL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL/action/storage_attestation","attest_author":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL/action/author_attestation","sign_citation":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL/action/citation_signature","submit_replication":"https://pith.science/pith/75C7JM4LZSNGDUADPRKIFG6BIL/action/replication_record"}},"created_at":"2026-07-05T11:50:44.089704+00:00","updated_at":"2026-07-05T11:50:44.089704+00:00"}