{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:24I6DOP73JTXHPTUQ4GVMMYRPM","short_pith_number":"pith:24I6DOP7","schema_version":"1.0","canonical_sha256":"d711e1b9ffda6773be74870d5633117b3c4893e2e07e84c6f289606f4399c853","source":{"kind":"arxiv","id":"2405.12609","version":6},"attestation_state":"computed","paper":{"title":"Mamba in Speech: Towards an Alternative to Self-Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Beena Ahmed, Eliathamby Ambikairajah, Haizhou Li, Hexin Liu, Julien Epps, Qiquan Zhang, Tianyi Xiao, Xiangyu Zhang, Xinyuan Qian","submitted_at":"2024-05-21T09:04:48Z","abstract_excerpt":"Transformer and its derivatives have achieved success in diverse tasks across computer vision, natural language processing, and speech processing. To reduce the complexity of computations within the multi-head self-attention mechanism in Transformer, Selective State Space Models (i.e., Mamba) were proposed as an alternative. Mamba exhibited its effectiveness in natural language processing and computer vision tasks, but its superiority has rarely been investigated in speech signal processing. This paper explores solutions for applying Mamba to speech processing by discussing two typical speech "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12609","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2024-05-21T09:04:48Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"1df2f764ec797449f9a2900f81703bdfbc09e1ef637c2194ff9a2bf33b0eb7a2","abstract_canon_sha256":"91daaece520bfa978837348b1e99c7ce51ec5ce912044af726728b88333481f2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:33.617607Z","signature_b64":"c6qNC6dlBHN/IN9lMO0VeP1d/EYAIleHtC+2ekXZBhVvvryesYtA5lwKvSqf14cxaZvuNSjR1KToIjs4BkLUDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d711e1b9ffda6773be74870d5633117b3c4893e2e07e84c6f289606f4399c853","last_reissued_at":"2026-07-05T10:54:33.617103Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:33.617103Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mamba in Speech: Towards an Alternative to Self-Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Beena Ahmed, Eliathamby Ambikairajah, Haizhou Li, Hexin Liu, Julien Epps, Qiquan Zhang, Tianyi Xiao, Xiangyu Zhang, Xinyuan Qian","submitted_at":"2024-05-21T09:04:48Z","abstract_excerpt":"Transformer and its derivatives have achieved success in diverse tasks across computer vision, natural language processing, and speech processing. To reduce the complexity of computations within the multi-head self-attention mechanism in Transformer, Selective State Space Models (i.e., Mamba) were proposed as an alternative. Mamba exhibited its effectiveness in natural language processing and computer vision tasks, but its superiority has rarely been investigated in speech signal processing. This paper explores solutions for applying Mamba to speech processing by discussing two typical speech "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12609","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12609/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12609","created_at":"2026-07-05T10:54:33.617162+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12609v6","created_at":"2026-07-05T10:54:33.617162+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12609","created_at":"2026-07-05T10:54:33.617162+00:00"},{"alias_kind":"pith_short_12","alias_value":"24I6DOP73JTX","created_at":"2026-07-05T10:54:33.617162+00:00"},{"alias_kind":"pith_short_16","alias_value":"24I6DOP73JTXHPTU","created_at":"2026-07-05T10:54:33.617162+00:00"},{"alias_kind":"pith_short_8","alias_value":"24I6DOP7","created_at":"2026-07-05T10:54:33.617162+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2408.01129","citing_title":"A Survey of Mamba","ref_index":233,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12606","citing_title":"An Exploration of Mamba for Speech Self-Supervised Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03563","citing_title":"State Space Models for Bioacoustics: A Comparative Evaluation with Transformers","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM","json":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM.json","graph_json":"https://pith.science/api/pith-number/24I6DOP73JTXHPTUQ4GVMMYRPM/graph.json","events_json":"https://pith.science/api/pith-number/24I6DOP73JTXHPTUQ4GVMMYRPM/events.json","paper":"https://pith.science/paper/24I6DOP7"},"agent_actions":{"view_html":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM","download_json":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM.json","view_paper":"https://pith.science/paper/24I6DOP7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12609&json=true","fetch_graph":"https://pith.science/api/pith-number/24I6DOP73JTXHPTUQ4GVMMYRPM/graph.json","fetch_events":"https://pith.science/api/pith-number/24I6DOP73JTXHPTUQ4GVMMYRPM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM/action/storage_attestation","attest_author":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM/action/author_attestation","sign_citation":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM/action/citation_signature","submit_replication":"https://pith.science/pith/24I6DOP73JTXHPTUQ4GVMMYRPM/action/replication_record"}},"created_at":"2026-07-05T10:54:33.617162+00:00","updated_at":"2026-07-05T10:54:33.617162+00:00"}