{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:BSEQP74TFPKULWZWGE7FJ6QOGE","short_pith_number":"pith:BSEQP74T","schema_version":"1.0","canonical_sha256":"0c8907ff932bd545db36313e54fa0e3125702007e8191b3c80517658ddec06d3","source":{"kind":"arxiv","id":"2212.00500","version":1},"attestation_state":"computed","paper":{"title":"MMSpeech: Multi-modal Multi-task Encoder-Decoder Pre-training for Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Chang Zhou, Jiaming Wang, Jingren Zhou, Shiliang Zhang, Xiaohuan Zhou, Zeyu Cui, Zhijie Yan","submitted_at":"2022-11-29T13:16:09Z","abstract_excerpt":"In this paper, we propose a novel multi-modal multi-task encoder-decoder pre-training framework (MMSpeech) for Mandarin automatic speech recognition (ASR), which employs both unlabeled speech and text data. The main difficulty in speech-text joint pre-training comes from the significant difference between speech and text modalities, especially for Mandarin speech and text. Unlike English and other languages with an alphabetic writing system, Mandarin uses an ideographic writing system where character and sound are not tightly mapped to one another. Therefore, we propose to introduce the phonem"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.00500","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MM","submitted_at":"2022-11-29T13:16:09Z","cross_cats_sorted":["cs.CL","cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"91f813a2369ae7b7ea0a81b269acb7688023703cad7afac5923a64198e4bbc48","abstract_canon_sha256":"6829f2d295e3a0c72fac3489701507dd8a35b704c510c7abee37c34f194f0879"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:21:34.805625Z","signature_b64":"pPPAGW/RKXdQIq0C6a0o4t4UW4oox2+rDAKVkRkdVX1L953CL94EI8Fv0MlymTSjiXhZz0OcT4xKzF8I3uecBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c8907ff932bd545db36313e54fa0e3125702007e8191b3c80517658ddec06d3","last_reissued_at":"2026-07-05T05:21:34.805258Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:21:34.805258Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMSpeech: Multi-modal Multi-task Encoder-Decoder Pre-training for Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Chang Zhou, Jiaming Wang, Jingren Zhou, Shiliang Zhang, Xiaohuan Zhou, Zeyu Cui, Zhijie Yan","submitted_at":"2022-11-29T13:16:09Z","abstract_excerpt":"In this paper, we propose a novel multi-modal multi-task encoder-decoder pre-training framework (MMSpeech) for Mandarin automatic speech recognition (ASR), which employs both unlabeled speech and text data. The main difficulty in speech-text joint pre-training comes from the significant difference between speech and text modalities, especially for Mandarin speech and text. Unlike English and other languages with an alphabetic writing system, Mandarin uses an ideographic writing system where character and sound are not tightly mapped to one another. Therefore, we propose to introduce the phonem"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.00500","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.00500/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.00500","created_at":"2026-07-05T05:21:34.805325+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.00500v1","created_at":"2026-07-05T05:21:34.805325+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.00500","created_at":"2026-07-05T05:21:34.805325+00:00"},{"alias_kind":"pith_short_12","alias_value":"BSEQP74TFPKU","created_at":"2026-07-05T05:21:34.805325+00:00"},{"alias_kind":"pith_short_16","alias_value":"BSEQP74TFPKULWZW","created_at":"2026-07-05T05:21:34.805325+00:00"},{"alias_kind":"pith_short_8","alias_value":"BSEQP74T","created_at":"2026-07-05T05:21:34.805325+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2407.10759","citing_title":"Qwen2-Audio Technical Report","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE","json":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE.json","graph_json":"https://pith.science/api/pith-number/BSEQP74TFPKULWZWGE7FJ6QOGE/graph.json","events_json":"https://pith.science/api/pith-number/BSEQP74TFPKULWZWGE7FJ6QOGE/events.json","paper":"https://pith.science/paper/BSEQP74T"},"agent_actions":{"view_html":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE","download_json":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE.json","view_paper":"https://pith.science/paper/BSEQP74T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.00500&json=true","fetch_graph":"https://pith.science/api/pith-number/BSEQP74TFPKULWZWGE7FJ6QOGE/graph.json","fetch_events":"https://pith.science/api/pith-number/BSEQP74TFPKULWZWGE7FJ6QOGE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE/action/storage_attestation","attest_author":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE/action/author_attestation","sign_citation":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE/action/citation_signature","submit_replication":"https://pith.science/pith/BSEQP74TFPKULWZWGE7FJ6QOGE/action/replication_record"}},"created_at":"2026-07-05T05:21:34.805325+00:00","updated_at":"2026-07-05T05:21:34.805325+00:00"}