{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UTRTXUT7V5WXIJA33TLTRDYQ3U","short_pith_number":"pith:UTRTXUT7","schema_version":"1.0","canonical_sha256":"a4e33bd27faf6d74241bdcd7388f10dd02bf4319050f0dc63a6a1d9f78083059","source":{"kind":"arxiv","id":"2504.16053","version":1},"attestation_state":"computed","paper":{"title":"LongMamba: Enhancing Mamba's Long Context Capabilities via Training-Free Receptive Field Enlargement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jan Kautz, Jihoon Hong, Kejing Xia, Pavlo Molchanov, Shizhe Diao, Xiangchi Yuan, Xin Dong, Yingyan Celine Lin, Yonggan Fu, Zhifan Ye","submitted_at":"2025-04-22T17:30:36Z","abstract_excerpt":"State space models (SSMs) have emerged as an efficient alternative to Transformer models for language modeling, offering linear computational complexity and constant memory usage as context length increases. However, despite their efficiency in handling long contexts, recent studies have shown that SSMs, such as Mamba models, generally underperform compared to Transformers in long-context understanding tasks. To address this significant shortfall and achieve both efficient and accurate long-context understanding, we propose LongMamba, a training-free technique that significantly enhances the l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.16053","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-22T17:30:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f7466956572abe4afcb01117d85f3af5ed1db2c7181d5eac402d02cdf73cb7a9","abstract_canon_sha256":"a9c3dac38378286d40b9e3421ec98cfe28875dd767bccc222db2691cbbbfb469"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:34.516265Z","signature_b64":"4HnI6GV+xBjrP5AhG2Wpj7LWe8IyEdNDeq5TMsDcoV/M5Wy4ywijo1lYeTU9iAb7tERHxMU762MwT5Z3l4+HCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a4e33bd27faf6d74241bdcd7388f10dd02bf4319050f0dc63a6a1d9f78083059","last_reissued_at":"2026-07-05T10:52:34.515760Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:34.515760Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LongMamba: Enhancing Mamba's Long Context Capabilities via Training-Free Receptive Field Enlargement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jan Kautz, Jihoon Hong, Kejing Xia, Pavlo Molchanov, Shizhe Diao, Xiangchi Yuan, Xin Dong, Yingyan Celine Lin, Yonggan Fu, Zhifan Ye","submitted_at":"2025-04-22T17:30:36Z","abstract_excerpt":"State space models (SSMs) have emerged as an efficient alternative to Transformer models for language modeling, offering linear computational complexity and constant memory usage as context length increases. However, despite their efficiency in handling long contexts, recent studies have shown that SSMs, such as Mamba models, generally underperform compared to Transformers in long-context understanding tasks. To address this significant shortfall and achieve both efficient and accurate long-context understanding, we propose LongMamba, a training-free technique that significantly enhances the l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.16053","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.16053/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.16053","created_at":"2026-07-05T10:52:34.515821+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.16053v1","created_at":"2026-07-05T10:52:34.515821+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.16053","created_at":"2026-07-05T10:52:34.515821+00:00"},{"alias_kind":"pith_short_12","alias_value":"UTRTXUT7V5WX","created_at":"2026-07-05T10:52:34.515821+00:00"},{"alias_kind":"pith_short_16","alias_value":"UTRTXUT7V5WXIJA3","created_at":"2026-07-05T10:52:34.515821+00:00"},{"alias_kind":"pith_short_8","alias_value":"UTRTXUT7","created_at":"2026-07-05T10:52:34.515821+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.26645","citing_title":"TTT3R: 3D Reconstruction as Test-Time Training","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20981","citing_title":"Echoes Over Time: Unlocking Length Generalization in Video-to-Audio Generation Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07658","citing_title":"Optimal Decay Spectra for Linear Recurrences","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U","json":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U.json","graph_json":"https://pith.science/api/pith-number/UTRTXUT7V5WXIJA33TLTRDYQ3U/graph.json","events_json":"https://pith.science/api/pith-number/UTRTXUT7V5WXIJA33TLTRDYQ3U/events.json","paper":"https://pith.science/paper/UTRTXUT7"},"agent_actions":{"view_html":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U","download_json":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U.json","view_paper":"https://pith.science/paper/UTRTXUT7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.16053&json=true","fetch_graph":"https://pith.science/api/pith-number/UTRTXUT7V5WXIJA33TLTRDYQ3U/graph.json","fetch_events":"https://pith.science/api/pith-number/UTRTXUT7V5WXIJA33TLTRDYQ3U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U/action/storage_attestation","attest_author":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U/action/author_attestation","sign_citation":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U/action/citation_signature","submit_replication":"https://pith.science/pith/UTRTXUT7V5WXIJA33TLTRDYQ3U/action/replication_record"}},"created_at":"2026-07-05T10:52:34.515821+00:00","updated_at":"2026-07-05T10:52:34.515821+00:00"}