{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RMYEUJESASORVS27T36HY7LPS4","short_pith_number":"pith:RMYEUJES","schema_version":"1.0","canonical_sha256":"8b304a2492049d1acb5f9efc7c7d6f9714a7745de2eac4180595c9f9472fa536","source":{"kind":"arxiv","id":"2412.01145","version":2},"attestation_state":"computed","paper":{"title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Bo Ren, Jinyu Li, Ruchao Fan, Rui Zhao, Shujie Liu, Yuxuan Hu","submitted_at":"2024-12-02T05:42:33Z","abstract_excerpt":"Integrating speech into LLM (speech-LLM) has gaining increased attention recently. The mainstream solution is to connect a well-trained speech encoder and LLM with a neural adapter. However, the length mismatch between the speech and text sequences are not well handled, leading to imperfect modality matching between the speech and text. In this work, we propose a novel neural adapter, AlignFormer, to reduce the length gap between the two modalities. AlignFormer consists of CTC and dynamic-window QFormer layers, where the CTC alignment provides the dynamic window information for QFormer. The LL"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.01145","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-12-02T05:42:33Z","cross_cats_sorted":[],"title_canon_sha256":"543ba1f49c356c13104e6e9ca62943bcce8858f168ee5d53006fbb66209b9c2b","abstract_canon_sha256":"731ccc28532968efbffd64cca9238cb78de6347296c0486c133dec1872bd0729"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:41.688489Z","signature_b64":"qkmfunynWnEehAA6h0Va2G/5SqrvtcxqcYeo4QJR6zxgokNl2RpBs2s+KBDCrAGszu9SV2kU8c/qvtWTn9wgBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b304a2492049d1acb5f9efc7c7d6f9714a7745de2eac4180595c9f9472fa536","last_reissued_at":"2026-07-05T11:31:41.687983Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:41.687983Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Bo Ren, Jinyu Li, Ruchao Fan, Rui Zhao, Shujie Liu, Yuxuan Hu","submitted_at":"2024-12-02T05:42:33Z","abstract_excerpt":"Integrating speech into LLM (speech-LLM) has gaining increased attention recently. The mainstream solution is to connect a well-trained speech encoder and LLM with a neural adapter. However, the length mismatch between the speech and text sequences are not well handled, leading to imperfect modality matching between the speech and text. In this work, we propose a novel neural adapter, AlignFormer, to reduce the length gap between the two modalities. AlignFormer consists of CTC and dynamic-window QFormer layers, where the CTC alignment provides the dynamic window information for QFormer. The LL"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.01145","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.01145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.01145","created_at":"2026-07-05T11:31:41.688042+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.01145v2","created_at":"2026-07-05T11:31:41.688042+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.01145","created_at":"2026-07-05T11:31:41.688042+00:00"},{"alias_kind":"pith_short_12","alias_value":"RMYEUJESASOR","created_at":"2026-07-05T11:31:41.688042+00:00"},{"alias_kind":"pith_short_16","alias_value":"RMYEUJESASORVS27","created_at":"2026-07-05T11:31:41.688042+00:00"},{"alias_kind":"pith_short_8","alias_value":"RMYEUJES","created_at":"2026-07-05T11:31:41.688042+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.08528","citing_title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4","json":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4.json","graph_json":"https://pith.science/api/pith-number/RMYEUJESASORVS27T36HY7LPS4/graph.json","events_json":"https://pith.science/api/pith-number/RMYEUJESASORVS27T36HY7LPS4/events.json","paper":"https://pith.science/paper/RMYEUJES"},"agent_actions":{"view_html":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4","download_json":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4.json","view_paper":"https://pith.science/paper/RMYEUJES","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.01145&json=true","fetch_graph":"https://pith.science/api/pith-number/RMYEUJESASORVS27T36HY7LPS4/graph.json","fetch_events":"https://pith.science/api/pith-number/RMYEUJESASORVS27T36HY7LPS4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4/action/storage_attestation","attest_author":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4/action/author_attestation","sign_citation":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4/action/citation_signature","submit_replication":"https://pith.science/pith/RMYEUJESASORVS27T36HY7LPS4/action/replication_record"}},"created_at":"2026-07-05T11:31:41.688042+00:00","updated_at":"2026-07-05T11:31:41.688042+00:00"}