{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:6LWUQYDTNHUXIQJ6CAJ5FI2HBV","short_pith_number":"pith:6LWUQYDT","schema_version":"1.0","canonical_sha256":"f2ed48607369e974413e1013d2a3470d50de589827187645d24b262ca6df2107","source":{"kind":"arxiv","id":"2206.08317","version":3},"attestation_state":"computed","paper":{"title":"Paraformer: Fast and Accurate Parallel Transformer for Non-autoregressive End-to-End Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ian McLoughlin, Shiliang Zhang, Zhifu Gao, Zhijie Yan","submitted_at":"2022-06-16T17:24:14Z","abstract_excerpt":"Transformers have recently dominated the ASR field. Although able to yield good performance, they involve an autoregressive (AR) decoder to generate tokens one by one, which is computationally inefficient. To speed up inference, non-autoregressive (NAR) methods, e.g. single-step NAR, were designed, to enable parallel generation. However, due to an independence assumption within the output tokens, performance of single-step NAR is inferior to that of AR models, especially with a large-scale corpus. There are two challenges to improving single-step NAR: Firstly to accurately predict the number o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.08317","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2022-06-16T17:24:14Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"7a388cc6c8f1a00eac3370cfbb7b1a960a23deb12538b7291819d6b57ae6ddb8","abstract_canon_sha256":"16f214fb86eef75f673c69edc1d096a1efac81cf3ff1b2f61a946b6ed8e3f0ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:56:14.153594Z","signature_b64":"Dd7fb5Bho/p43kBdGLyGLKWWlM6y5nNDyiFJW1LAO/rfmlI29e9K2DJBDA8da5FhWuu/yzS4skoNo/OwL+XrCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2ed48607369e974413e1013d2a3470d50de589827187645d24b262ca6df2107","last_reissued_at":"2026-07-05T05:56:14.153179Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:56:14.153179Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Paraformer: Fast and Accurate Parallel Transformer for Non-autoregressive End-to-End Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Ian McLoughlin, Shiliang Zhang, Zhifu Gao, Zhijie Yan","submitted_at":"2022-06-16T17:24:14Z","abstract_excerpt":"Transformers have recently dominated the ASR field. Although able to yield good performance, they involve an autoregressive (AR) decoder to generate tokens one by one, which is computationally inefficient. To speed up inference, non-autoregressive (NAR) methods, e.g. single-step NAR, were designed, to enable parallel generation. However, due to an independence assumption within the output tokens, performance of single-step NAR is inferior to that of AR models, especially with a large-scale corpus. There are two challenges to improving single-step NAR: Firstly to accurately predict the number o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.08317","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.08317/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.08317","created_at":"2026-07-05T05:56:14.153230+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.08317v3","created_at":"2026-07-05T05:56:14.153230+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.08317","created_at":"2026-07-05T05:56:14.153230+00:00"},{"alias_kind":"pith_short_12","alias_value":"6LWUQYDTNHUX","created_at":"2026-07-05T05:56:14.153230+00:00"},{"alias_kind":"pith_short_16","alias_value":"6LWUQYDTNHUXIQJ6","created_at":"2026-07-05T05:56:14.153230+00:00"},{"alias_kind":"pith_short_8","alias_value":"6LWUQYDT","created_at":"2026-07-05T05:56:14.153230+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01804","citing_title":"SpeechEditBench: A Bilingual Multi-Attribute Benchmark for Instruction-Guided Speech Editing","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18512","citing_title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2502.11946","citing_title":"Step-Audio: Unified Understanding and Generation in Intelligent Speech Interaction","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2511.09282","citing_title":"End-to-end Contrastive Language-Speech Pretraining Model For Long-form Spoken Question Answering","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12783","citing_title":"SQuTR: A Robustness Benchmark for Spoken Query to Text Retrieval under Acoustic Noise","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13073","citing_title":"OmniTrace: A Unified Framework for Generation-Time Attribution in Omni-Modal LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01897","citing_title":"FastTurn: Unifying Acoustic and Streaming Semantic Cues for Low-Latency and Robust Turn Detection","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10027","citing_title":"Speech-based Psychological Crisis Assessment using LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2504.18425","citing_title":"Kimi-Audio Technical Report","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21406","citing_title":"Full-Duplex Interaction in Spoken Dialogue Systems: A Comprehensive Study from the ICASSP 2026 HumDial Challenge","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08000","citing_title":"PASK: Toward Intent-Aware Proactive Agents with Long-Term Memory","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06765","citing_title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV","json":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV.json","graph_json":"https://pith.science/api/pith-number/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/graph.json","events_json":"https://pith.science/api/pith-number/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/events.json","paper":"https://pith.science/paper/6LWUQYDT"},"agent_actions":{"view_html":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV","download_json":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV.json","view_paper":"https://pith.science/paper/6LWUQYDT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.08317&json=true","fetch_graph":"https://pith.science/api/pith-number/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/graph.json","fetch_events":"https://pith.science/api/pith-number/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/action/storage_attestation","attest_author":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/action/author_attestation","sign_citation":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/action/citation_signature","submit_replication":"https://pith.science/pith/6LWUQYDTNHUXIQJ6CAJ5FI2HBV/action/replication_record"}},"created_at":"2026-07-05T05:56:14.153230+00:00","updated_at":"2026-07-05T05:56:14.153230+00:00"}