{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E46JO6FMXP7RFIABN6ATHFW6N4","short_pith_number":"pith:E46JO6FM","schema_version":"1.0","canonical_sha256":"273c9778acbbff12a0016f813396de6f1809a12783853f306a7376bb30598810","source":{"kind":"arxiv","id":"2408.17175","version":3},"attestation_state":"computed","paper":{"title":"Codec Does Matter: Exploring the Semantic Shortcoming of Codec for Audio Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Hongzhan Lin, Jiahao Pan, Jiahe Lei, Jianyi Chen, Peiwen Sun, Qifeng Liu, Qiuqiang Kong, Wei Xue, Xu Tan, Yike Guo, Zhen Ye, Zheqi Dai","submitted_at":"2024-08-30T10:24:07Z","abstract_excerpt":"Recent advancements in audio generation have been significantly propelled by the capabilities of Large Language Models (LLMs). The existing research on audio LLM has primarily focused on enhancing the architecture and scale of audio language models, as well as leveraging larger datasets, and generally, acoustic codecs, such as EnCodec, are used for audio tokenization. However, these codecs were originally designed for audio compression, which may lead to suboptimal performance in the context of audio LLM. Our research aims to address the shortcomings of current audio LLM codecs, particularly t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.17175","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2024-08-30T10:24:07Z","cross_cats_sorted":["cs.AI","cs.CL","cs.SD"],"title_canon_sha256":"722aaddfa78b144c8abb5bead7711f3f049de8aff744eacb88b2e4d2a0ff2fd4","abstract_canon_sha256":"07096c177004914faa227ce8e6e00fc5c64e75380ca4da5916465122fa63a5f4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:01.279812Z","signature_b64":"NNEPmOxJcpyg196+4PK+3RxULAqlr2J50FZQ1VASqvIgv2hu32Lkti6BgqFBDvtPdXGzfvf73UriK4lqexYqBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"273c9778acbbff12a0016f813396de6f1809a12783853f306a7376bb30598810","last_reissued_at":"2026-07-05T09:41:01.279367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:01.279367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Codec Does Matter: Exploring the Semantic Shortcoming of Codec for Audio Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Hongzhan Lin, Jiahao Pan, Jiahe Lei, Jianyi Chen, Peiwen Sun, Qifeng Liu, Qiuqiang Kong, Wei Xue, Xu Tan, Yike Guo, Zhen Ye, Zheqi Dai","submitted_at":"2024-08-30T10:24:07Z","abstract_excerpt":"Recent advancements in audio generation have been significantly propelled by the capabilities of Large Language Models (LLMs). The existing research on audio LLM has primarily focused on enhancing the architecture and scale of audio language models, as well as leveraging larger datasets, and generally, acoustic codecs, such as EnCodec, are used for audio tokenization. However, these codecs were originally designed for audio compression, which may lead to suboptimal performance in the context of audio LLM. Our research aims to address the shortcomings of current audio LLM codecs, particularly t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.17175","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.17175/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.17175","created_at":"2026-07-05T09:41:01.279421+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.17175v3","created_at":"2026-07-05T09:41:01.279421+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.17175","created_at":"2026-07-05T09:41:01.279421+00:00"},{"alias_kind":"pith_short_12","alias_value":"E46JO6FMXP7R","created_at":"2026-07-05T09:41:01.279421+00:00"},{"alias_kind":"pith_short_16","alias_value":"E46JO6FMXP7RFIAB","created_at":"2026-07-05T09:41:01.279421+00:00"},{"alias_kind":"pith_short_8","alias_value":"E46JO6FM","created_at":"2026-07-05T09:41:01.279421+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06357","citing_title":"F3-Tokenizer: Taming Audio Autoencoder Latents for Understanding and Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29480","citing_title":"DTM-Codec: Dynamic Token Masking for VFR Speech Coding with Efficient Boundary Selection","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01537","citing_title":"Two-Dimensional Quantization for Geometry-Aware Audio Coding","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27607","citing_title":"JaiTTS: A Thai Voice Cloning Model","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26296","citing_title":"SPG-Codec: Exploring the Role and Boundaries of Semantic Priors in Ultra-Low-Bitrate Neural Speech Coding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10281","citing_title":"Drum Synthesis from Expressive Drum Grids via Neural Audio Codecs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27607","citing_title":"JaiTTS: A Thai Voice Cloning Model","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4","json":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4.json","graph_json":"https://pith.science/api/pith-number/E46JO6FMXP7RFIABN6ATHFW6N4/graph.json","events_json":"https://pith.science/api/pith-number/E46JO6FMXP7RFIABN6ATHFW6N4/events.json","paper":"https://pith.science/paper/E46JO6FM"},"agent_actions":{"view_html":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4","download_json":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4.json","view_paper":"https://pith.science/paper/E46JO6FM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.17175&json=true","fetch_graph":"https://pith.science/api/pith-number/E46JO6FMXP7RFIABN6ATHFW6N4/graph.json","fetch_events":"https://pith.science/api/pith-number/E46JO6FMXP7RFIABN6ATHFW6N4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4/action/storage_attestation","attest_author":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4/action/author_attestation","sign_citation":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4/action/citation_signature","submit_replication":"https://pith.science/pith/E46JO6FMXP7RFIABN6ATHFW6N4/action/replication_record"}},"created_at":"2026-07-05T09:41:01.279421+00:00","updated_at":"2026-07-05T09:41:01.279421+00:00"}