{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UMREHQPHGBWIMS6ZPOVBWXZHVG","short_pith_number":"pith:UMREHQPH","schema_version":"1.0","canonical_sha256":"a32243c1e7306c864bd97baa1b5f27a9ab961df183cc73ca3aebcc9bab96f73a","source":{"kind":"arxiv","id":"2305.16342","version":2},"attestation_state":"computed","paper":{"title":"InterFormer: Interactive Local and Global Features Fusion for Automatic Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Feng Chen, Li-Fang Wei, Qi Liu, Song-Lu Chen, Tian-Hao Zhang, Xinyuan Qian, Xu-Cheng Yin, Zhi-Hao Lai","submitted_at":"2023-05-24T08:43:44Z","abstract_excerpt":"The local and global features are both essential for automatic speech recognition (ASR). Many recent methods have verified that simply combining local and global features can further promote ASR performance. However, these methods pay less attention to the interaction of local and global features, and their series architectures are rigid to reflect local and global relationships. To address these issues, this paper proposes InterFormer for interactive local and global features fusion to learn a better representation for ASR. Specifically, we combine the convolution block with the transformer b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16342","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T08:43:44Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"a6d79aa8578da61143cdde700b47da39043ab9a75706a6987067e2f6e2ad894a","abstract_canon_sha256":"d659f8b85a40da513a77d127a74a05cff218ced0b05edb60cd0da5264a0add2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:14:43.874091Z","signature_b64":"Hw6VEKVWuzKHekGgAoWBwmH9LAdnNZFHnT6VyTCU10gmMVUXIl383k1SMpgh8aygCWb5dc3BIsIe3WXvW7dWDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a32243c1e7306c864bd97baa1b5f27a9ab961df183cc73ca3aebcc9bab96f73a","last_reissued_at":"2026-07-05T06:14:43.873690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:14:43.873690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InterFormer: Interactive Local and Global Features Fusion for Automatic Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Feng Chen, Li-Fang Wei, Qi Liu, Song-Lu Chen, Tian-Hao Zhang, Xinyuan Qian, Xu-Cheng Yin, Zhi-Hao Lai","submitted_at":"2023-05-24T08:43:44Z","abstract_excerpt":"The local and global features are both essential for automatic speech recognition (ASR). Many recent methods have verified that simply combining local and global features can further promote ASR performance. However, these methods pay less attention to the interaction of local and global features, and their series architectures are rigid to reflect local and global relationships. To address these issues, this paper proposes InterFormer for interactive local and global features fusion to learn a better representation for ASR. Specifically, we combine the convolution block with the transformer b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16342","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16342/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16342","created_at":"2026-07-05T06:14:43.873745+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16342v2","created_at":"2026-07-05T06:14:43.873745+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16342","created_at":"2026-07-05T06:14:43.873745+00:00"},{"alias_kind":"pith_short_12","alias_value":"UMREHQPHGBWI","created_at":"2026-07-05T06:14:43.873745+00:00"},{"alias_kind":"pith_short_16","alias_value":"UMREHQPHGBWIMS6Z","created_at":"2026-07-05T06:14:43.873745+00:00"},{"alias_kind":"pith_short_8","alias_value":"UMREHQPH","created_at":"2026-07-05T06:14:43.873745+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.24301","citing_title":"Category-aware EEG image generation based on wavelet transform and contrast semantic loss","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG","json":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG.json","graph_json":"https://pith.science/api/pith-number/UMREHQPHGBWIMS6ZPOVBWXZHVG/graph.json","events_json":"https://pith.science/api/pith-number/UMREHQPHGBWIMS6ZPOVBWXZHVG/events.json","paper":"https://pith.science/paper/UMREHQPH"},"agent_actions":{"view_html":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG","download_json":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG.json","view_paper":"https://pith.science/paper/UMREHQPH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16342&json=true","fetch_graph":"https://pith.science/api/pith-number/UMREHQPHGBWIMS6ZPOVBWXZHVG/graph.json","fetch_events":"https://pith.science/api/pith-number/UMREHQPHGBWIMS6ZPOVBWXZHVG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG/action/storage_attestation","attest_author":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG/action/author_attestation","sign_citation":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG/action/citation_signature","submit_replication":"https://pith.science/pith/UMREHQPHGBWIMS6ZPOVBWXZHVG/action/replication_record"}},"created_at":"2026-07-05T06:14:43.873745+00:00","updated_at":"2026-07-05T06:14:43.873745+00:00"}