{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3S6VQLWYOXUDBCZIDQINDHLNCO","short_pith_number":"pith:3S6VQLWY","schema_version":"1.0","canonical_sha256":"dcbd582ed875e8308b281c10d19d6d1383b352a94e4a24ccdcce1709f7bb4a5d","source":{"kind":"arxiv","id":"2406.11037","version":1},"attestation_state":"computed","paper":{"title":"NAST: Noise Aware Speech Tokenization for Speech Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Shoval Messica, Yossi Adi","submitted_at":"2024-06-16T18:20:45Z","abstract_excerpt":"Speech tokenization is the task of representing speech signals as a sequence of discrete units. Such representations can be later used for various downstream tasks including automatic speech recognition, text-to-speech, etc. More relevant to this study, such representation serves as the basis of Speech Language Models. In this work, we tackle the task of speech tokenization under the noisy setup and present NAST: Noise Aware Speech Tokenization for Speech Language Models. NAST is composed of three main components: (i) a predictor; (ii) a residual encoder; and (iii) a decoder. We evaluate the e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11037","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-06-16T18:20:45Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"449cdea13861919621f59d1cb6c5000cb3e41106a8a90dd5759cb53c71e9ac8a","abstract_canon_sha256":"438f837d18ca49aeb6e671dea9e47e5fda031857df29fc38872bd64aa366b54e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:47.974219Z","signature_b64":"tZEcYC2UpiRcx/3xbDDU5efAe0CIW2Hx5INHvm3xHni6aZahpd4qDQCmZV954eQ2tHpq89UmOtuB1TJvy+E5CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dcbd582ed875e8308b281c10d19d6d1383b352a94e4a24ccdcce1709f7bb4a5d","last_reissued_at":"2026-07-05T08:32:47.973765Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:47.973765Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NAST: Noise Aware Speech Tokenization for Speech Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Shoval Messica, Yossi Adi","submitted_at":"2024-06-16T18:20:45Z","abstract_excerpt":"Speech tokenization is the task of representing speech signals as a sequence of discrete units. Such representations can be later used for various downstream tasks including automatic speech recognition, text-to-speech, etc. More relevant to this study, such representation serves as the basis of Speech Language Models. In this work, we tackle the task of speech tokenization under the noisy setup and present NAST: Noise Aware Speech Tokenization for Speech Language Models. NAST is composed of three main components: (i) a predictor; (ii) a residual encoder; and (iii) a decoder. We evaluate the e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11037","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11037/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11037","created_at":"2026-07-05T08:32:47.973828+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11037v1","created_at":"2026-07-05T08:32:47.973828+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11037","created_at":"2026-07-05T08:32:47.973828+00:00"},{"alias_kind":"pith_short_12","alias_value":"3S6VQLWYOXUD","created_at":"2026-07-05T08:32:47.973828+00:00"},{"alias_kind":"pith_short_16","alias_value":"3S6VQLWYOXUDBCZI","created_at":"2026-07-05T08:32:47.973828+00:00"},{"alias_kind":"pith_short_8","alias_value":"3S6VQLWY","created_at":"2026-07-05T08:32:47.973828+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12940","citing_title":"Self-Guidance: Enhancing Neural Codecs via Decoder Manifold Alignment","ref_index":80,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO","json":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO.json","graph_json":"https://pith.science/api/pith-number/3S6VQLWYOXUDBCZIDQINDHLNCO/graph.json","events_json":"https://pith.science/api/pith-number/3S6VQLWYOXUDBCZIDQINDHLNCO/events.json","paper":"https://pith.science/paper/3S6VQLWY"},"agent_actions":{"view_html":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO","download_json":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO.json","view_paper":"https://pith.science/paper/3S6VQLWY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11037&json=true","fetch_graph":"https://pith.science/api/pith-number/3S6VQLWYOXUDBCZIDQINDHLNCO/graph.json","fetch_events":"https://pith.science/api/pith-number/3S6VQLWYOXUDBCZIDQINDHLNCO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO/action/storage_attestation","attest_author":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO/action/author_attestation","sign_citation":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO/action/citation_signature","submit_replication":"https://pith.science/pith/3S6VQLWYOXUDBCZIDQINDHLNCO/action/replication_record"}},"created_at":"2026-07-05T08:32:47.973828+00:00","updated_at":"2026-07-05T08:32:47.973828+00:00"}