{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F2W7UB56HA52JQ43PO4SC4C3O5","short_pith_number":"pith:F2W7UB56","schema_version":"1.0","canonical_sha256":"2eadfa07be383ba4c39b7bb921705b775ca77155c07ca46ce13f3d4d4d9fcfc9","source":{"kind":"arxiv","id":"2505.17076","version":3},"attestation_state":"computed","paper":{"title":"Impact of Frame Rates on Speech Tokenizer: A Case Study on Mandarin and English","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Eng Siong Chng, Fei Tian, Haoyang Zhang, Hexin Liu, Junqi Zhao, Leibny Paola Garcia, Qiquan Zhang, Xiangyu Zhang, Xuerui Yang, Yuchen Hu","submitted_at":"2025-05-20T06:01:19Z","abstract_excerpt":"The speech tokenizer plays a crucial role in recent speech tasks, generally serving as a bridge between speech signals and language models. While low-frame-rate codecs are widely employed as speech tokenizers, the impact of frame rates on speech tokens remains underexplored. In this study, we investigate how varying frame rates affect speech tokenization by examining Mandarin and English, two typologically distinct languages. We encode speech at different frame rates and evaluate the resulting semantic tokens in the speech recognition task. Our findings reveal that frame rate variations influe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17076","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-20T06:01:19Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"881dfc6ef964f4e60ece8a6f7f058abed3aa2369ba2b5db4935dfb9f33530933","abstract_canon_sha256":"4c586a8e11985b21e9f3f3456d6f7938dc8935610411e1d7f31aa88e3a324cf7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:54.393393Z","signature_b64":"mdTeBpM1+j4YTWMlPxGZcFqWCMdA/I4+UTe1Ppia1fPs2sC4zG68quPjtceX1NP29M1xMLJ19ti1CZk402+JCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2eadfa07be383ba4c39b7bb921705b775ca77155c07ca46ce13f3d4d4d9fcfc9","last_reissued_at":"2026-07-05T11:20:54.392865Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:54.392865Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Impact of Frame Rates on Speech Tokenizer: A Case Study on Mandarin and English","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Eng Siong Chng, Fei Tian, Haoyang Zhang, Hexin Liu, Junqi Zhao, Leibny Paola Garcia, Qiquan Zhang, Xiangyu Zhang, Xuerui Yang, Yuchen Hu","submitted_at":"2025-05-20T06:01:19Z","abstract_excerpt":"The speech tokenizer plays a crucial role in recent speech tasks, generally serving as a bridge between speech signals and language models. While low-frame-rate codecs are widely employed as speech tokenizers, the impact of frame rates on speech tokens remains underexplored. In this study, we investigate how varying frame rates affect speech tokenization by examining Mandarin and English, two typologically distinct languages. We encode speech at different frame rates and evaluate the resulting semantic tokens in the speech recognition task. Our findings reveal that frame rate variations influe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17076","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17076/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17076","created_at":"2026-07-05T11:20:54.392937+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17076v3","created_at":"2026-07-05T11:20:54.392937+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17076","created_at":"2026-07-05T11:20:54.392937+00:00"},{"alias_kind":"pith_short_12","alias_value":"F2W7UB56HA52","created_at":"2026-07-05T11:20:54.392937+00:00"},{"alias_kind":"pith_short_16","alias_value":"F2W7UB56HA52JQ43","created_at":"2026-07-05T11:20:54.392937+00:00"},{"alias_kind":"pith_short_8","alias_value":"F2W7UB56","created_at":"2026-07-05T11:20:54.392937+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5","json":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5.json","graph_json":"https://pith.science/api/pith-number/F2W7UB56HA52JQ43PO4SC4C3O5/graph.json","events_json":"https://pith.science/api/pith-number/F2W7UB56HA52JQ43PO4SC4C3O5/events.json","paper":"https://pith.science/paper/F2W7UB56"},"agent_actions":{"view_html":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5","download_json":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5.json","view_paper":"https://pith.science/paper/F2W7UB56","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17076&json=true","fetch_graph":"https://pith.science/api/pith-number/F2W7UB56HA52JQ43PO4SC4C3O5/graph.json","fetch_events":"https://pith.science/api/pith-number/F2W7UB56HA52JQ43PO4SC4C3O5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5/action/storage_attestation","attest_author":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5/action/author_attestation","sign_citation":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5/action/citation_signature","submit_replication":"https://pith.science/pith/F2W7UB56HA52JQ43PO4SC4C3O5/action/replication_record"}},"created_at":"2026-07-05T11:20:54.392937+00:00","updated_at":"2026-07-05T11:20:54.392937+00:00"}