{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3FPBJFFE5E37SVZOQK5BZX633L","short_pith_number":"pith:3FPBJFFE","schema_version":"1.0","canonical_sha256":"d95e1494a4e937f9572e82ba1cdfdbdac99c7a16a6761ac8b3f0dfc05dd9020c","source":{"kind":"arxiv","id":"2411.06968","version":1},"attestation_state":"computed","paper":{"title":"Mamba-based Decoder-Only Approach with Bidirectional Speech Modeling for Speech Recognition","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Koichi Miyazaki, Masato Murata, Yoshiki Masuyama","submitted_at":"2024-11-11T13:17:24Z","abstract_excerpt":"Selective state space models (SSMs) represented by Mamba have demonstrated their computational efficiency and promising outcomes in various tasks, including automatic speech recognition (ASR). Mamba has been applied to ASR task with the attention-based encoder-decoder framework, where the cross-attention mechanism between encoder and decoder remains. This paper explores the capability of Mamba as the decoder-only architecture in ASR task. Our MAmba-based DEcoder-ONly approach (MADEON) consists of a single decoder that takes speech tokens as a condition and predicts text tokens in an autoregres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.06968","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2024-11-11T13:17:24Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"a1e8025e0d1f6aa9b1f507db730e189ba1e10f33625e77fc69743e44051ac106","abstract_canon_sha256":"a7ae742d71c951405457dbd39b94c2cc8027ae9d29443676f13b7d18c805b0fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:54.592177Z","signature_b64":"uNmfqLUlDUO9WOv2xN471A4G0nLQz66FbEZjVPha11f7XGTXKXdNwsMo2pMzpni41PTq0+8J6upNkrIDogxzAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d95e1494a4e937f9572e82ba1cdfdbdac99c7a16a6761ac8b3f0dfc05dd9020c","last_reissued_at":"2026-07-05T09:33:54.591761Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:54.591761Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mamba-based Decoder-Only Approach with Bidirectional Speech Modeling for Speech Recognition","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Koichi Miyazaki, Masato Murata, Yoshiki Masuyama","submitted_at":"2024-11-11T13:17:24Z","abstract_excerpt":"Selective state space models (SSMs) represented by Mamba have demonstrated their computational efficiency and promising outcomes in various tasks, including automatic speech recognition (ASR). Mamba has been applied to ASR task with the attention-based encoder-decoder framework, where the cross-attention mechanism between encoder and decoder remains. This paper explores the capability of Mamba as the decoder-only architecture in ASR task. Our MAmba-based DEcoder-ONly approach (MADEON) consists of a single decoder that takes speech tokens as a condition and predicts text tokens in an autoregres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.06968","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.06968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.06968","created_at":"2026-07-05T09:33:54.591832+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.06968v1","created_at":"2026-07-05T09:33:54.591832+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.06968","created_at":"2026-07-05T09:33:54.591832+00:00"},{"alias_kind":"pith_short_12","alias_value":"3FPBJFFE5E37","created_at":"2026-07-05T09:33:54.591832+00:00"},{"alias_kind":"pith_short_16","alias_value":"3FPBJFFE5E37SVZO","created_at":"2026-07-05T09:33:54.591832+00:00"},{"alias_kind":"pith_short_8","alias_value":"3FPBJFFE","created_at":"2026-07-05T09:33:54.591832+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.19761","citing_title":"Accurate, fast, cheap: Choose three. Replacing Multi-Head-Attention with Bidirectional Recurrent Attention for Long-Form ASR","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L","json":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L.json","graph_json":"https://pith.science/api/pith-number/3FPBJFFE5E37SVZOQK5BZX633L/graph.json","events_json":"https://pith.science/api/pith-number/3FPBJFFE5E37SVZOQK5BZX633L/events.json","paper":"https://pith.science/paper/3FPBJFFE"},"agent_actions":{"view_html":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L","download_json":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L.json","view_paper":"https://pith.science/paper/3FPBJFFE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.06968&json=true","fetch_graph":"https://pith.science/api/pith-number/3FPBJFFE5E37SVZOQK5BZX633L/graph.json","fetch_events":"https://pith.science/api/pith-number/3FPBJFFE5E37SVZOQK5BZX633L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L/action/storage_attestation","attest_author":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L/action/author_attestation","sign_citation":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L/action/citation_signature","submit_replication":"https://pith.science/pith/3FPBJFFE5E37SVZOQK5BZX633L/action/replication_record"}},"created_at":"2026-07-05T09:33:54.591832+00:00","updated_at":"2026-07-05T09:33:54.591832+00:00"}