{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DUEPQEX4XTLU5ZBO2JPBQJGNSI","short_pith_number":"pith:DUEPQEX4","schema_version":"1.0","canonical_sha256":"1d08f812fcbcd74ee42ed25e1824cd922917617084851bad4d0f00a08bf3793d","source":{"kind":"arxiv","id":"2405.00233","version":2},"attestation_state":"computed","paper":{"title":"SemantiCodec: An Ultra Low Bitrate Semantic Audio Codec for General Sound","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MM","eess.AS","eess.SP"],"primary_cat":"cs.SD","authors_text":"Haohe Liu, Mark D. Plumbley, Mengyue Wu, Wenwu Wang, Xuenan Xu, Yi Yuan","submitted_at":"2024-04-30T22:51:36Z","abstract_excerpt":"Large language models (LLMs) have significantly advanced audio processing through audio codecs that convert audio into discrete tokens, enabling the application of language modelling techniques to audio data. However, traditional codecs often operate at high bitrates or within narrow domains such as speech and lack the semantic clues required for efficient language modelling. Addressing these challenges, we introduce SemantiCodec, a novel codec designed to compress audio into fewer than a hundred tokens per second across diverse audio types, including speech, general sound, and music, without "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.00233","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-04-30T22:51:36Z","cross_cats_sorted":["cs.AI","cs.MM","eess.AS","eess.SP"],"title_canon_sha256":"06905aca60edc67a63394b89d5d73e0fbfb1cab3e90f5df6ec4792bc4ca7ea7c","abstract_canon_sha256":"c1928e35e3513986d28d547b72fab9de650fdfabe23d065c752a5213045c93d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:34.216454Z","signature_b64":"srYaRIO1ZfzR41I/6FsxX2/teGD16FPwzcfh+heBxi3fL2vD1bSekTyaIfbHuTDZnak6F0750RcSE3jCCByfAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d08f812fcbcd74ee42ed25e1824cd922917617084851bad4d0f00a08bf3793d","last_reissued_at":"2026-07-05T09:41:34.215954Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:34.215954Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SemantiCodec: An Ultra Low Bitrate Semantic Audio Codec for General Sound","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MM","eess.AS","eess.SP"],"primary_cat":"cs.SD","authors_text":"Haohe Liu, Mark D. Plumbley, Mengyue Wu, Wenwu Wang, Xuenan Xu, Yi Yuan","submitted_at":"2024-04-30T22:51:36Z","abstract_excerpt":"Large language models (LLMs) have significantly advanced audio processing through audio codecs that convert audio into discrete tokens, enabling the application of language modelling techniques to audio data. However, traditional codecs often operate at high bitrates or within narrow domains such as speech and lack the semantic clues required for efficient language modelling. Addressing these challenges, we introduce SemantiCodec, a novel codec designed to compress audio into fewer than a hundred tokens per second across diverse audio types, including speech, general sound, and music, without "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00233","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.00233/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.00233","created_at":"2026-07-05T09:41:34.216020+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.00233v2","created_at":"2026-07-05T09:41:34.216020+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00233","created_at":"2026-07-05T09:41:34.216020+00:00"},{"alias_kind":"pith_short_12","alias_value":"DUEPQEX4XTLU","created_at":"2026-07-05T09:41:34.216020+00:00"},{"alias_kind":"pith_short_16","alias_value":"DUEPQEX4XTLU5ZBO","created_at":"2026-07-05T09:41:34.216020+00:00"},{"alias_kind":"pith_short_8","alias_value":"DUEPQEX4","created_at":"2026-07-05T09:41:34.216020+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21893","citing_title":"AugCodec: A Low-Bitrate Disentangled Neural Speech Codec via Data Augmentation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01537","citing_title":"Two-Dimensional Quantization for Geometry-Aware Audio Coding","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11192","citing_title":"Exploring Token-Space Manipulation in Latent Audio Tokenizers","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2410.00037","citing_title":"Moshi: a speech-text foundation model for real-time dialogue","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI","json":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI.json","graph_json":"https://pith.science/api/pith-number/DUEPQEX4XTLU5ZBO2JPBQJGNSI/graph.json","events_json":"https://pith.science/api/pith-number/DUEPQEX4XTLU5ZBO2JPBQJGNSI/events.json","paper":"https://pith.science/paper/DUEPQEX4"},"agent_actions":{"view_html":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI","download_json":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI.json","view_paper":"https://pith.science/paper/DUEPQEX4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.00233&json=true","fetch_graph":"https://pith.science/api/pith-number/DUEPQEX4XTLU5ZBO2JPBQJGNSI/graph.json","fetch_events":"https://pith.science/api/pith-number/DUEPQEX4XTLU5ZBO2JPBQJGNSI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI/action/storage_attestation","attest_author":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI/action/author_attestation","sign_citation":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI/action/citation_signature","submit_replication":"https://pith.science/pith/DUEPQEX4XTLU5ZBO2JPBQJGNSI/action/replication_record"}},"created_at":"2026-07-05T09:41:34.216020+00:00","updated_at":"2026-07-05T09:41:34.216020+00:00"}