{"as_of":"2026-08-12T15:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9214b2000111673ac4e4965b36a4e92fe7f3660bfed77637f2ae56585b29ed88","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T22:52:33.772825Z","state":"measured"},{"denominator":83,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":83,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:58:54.578841Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-11T04:25:56.146025Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00658","snapshot_observed_at":"2026-08-07T04:58:54.578841Z","title":"Understanding and mitigating bottlenecks of state space models through the lens of recency and over-smoothing","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09316","last_updated":"2025-06-17T04:56:46Z","snapshot_observed_at":"2026-08-07T11:02:36.129585Z","submitted_at":"2025-06-11T01:25:06Z","title":"On-the-Fly Adaptive Distillation of Transformer to Dual-State Linear Attention","version":3},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-07T04:58:54.578841Z"},"links":{"cited_paper":"/paper/2501.00658","citing_paper":"/paper/2506.09316"},"observation_digest":"sha256:ddc92129b319497e9443fb62ec76796f399f94bae5bb527cdbc510c69042dc7a","observation_id":"78136a9d-755a-4e54-a2d9-3a851978ff3c","resolution":{"observed_at":"2026-08-07T04:58:54.578841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"cited_work":{"arxiv_id":"2501.00658","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.00658","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"74475bdc-bb73-4e15-bf4f-4e1bc4fe6d87","year":2024},"citing_paper":{"arxiv_id":"2605.06997","last_updated":"2026-05-07T22:26:27Z","snapshot_observed_at":"2026-08-02T23:31:22.583542Z","submitted_at":"2026-05-07T22:26:27Z","title":"Echo: KV-Cache-Free Associative Recall with Spectral Koopman Operators","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-11T01:26:48.026538Z"},"links":{"cited_paper":"/paper/2501.00658","citing_paper":"/paper/2605.06997"},"observation_digest":"sha256:135e37b328dfb5b8af0afccef751eb79d445274cb02c2465bc3f18e0ad2f83e2","observation_id":"e6a00698-189c-4a82-abb3-014f0b22a923","resolution":{"observed_at":"2026-05-11T04:25:56.150123Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00658","snapshot_observed_at":"2026-08-02T06:39:25.045565Z","title":"Understanding and mitigating bottlenecks of state space models through the lens of recency and over-smoothing.arXiv preprint arXiv:2501.00658, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14144","last_updated":"2026-07-31T05:37:48Z","snapshot_observed_at":"2026-08-06T19:33:05.649205Z","submitted_at":"2026-07-14T05:52:12Z","title":"The Capability Convergence Hypothesis: Capability from Access Structure, Not Scale","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T06:39:25.045565Z"},"links":{"cited_paper":"/paper/2501.00658","citing_paper":"/paper/2607.14144"},"observation_digest":"sha256:60823a5fa0cca09d35134cb2cf6b7eddd5bbd6bde9eab09f22234a47d6701614","observation_id":"b59c1c1e-61d6-4d90-a7d9-14f50aeb86fb","resolution":{"observed_at":"2026-08-02T06:39:25.045565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00658","snapshot_observed_at":"2026-08-03T02:03:25.442040Z","title":"Understanding and mitigating bottlenecks of state space models through the lens of recency and over-smoothing.arXiv preprint arXiv:2501.00658, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14144","last_updated":"2026-07-31T05:37:48Z","snapshot_observed_at":"2026-08-06T19:33:05.649205Z","submitted_at":"2026-07-14T05:52:12Z","title":"The Capability Convergence Hypothesis: Capability from Access Structure, Not Scale","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-03T02:03:25.442040Z"},"links":{"cited_paper":"/paper/2501.00658","citing_paper":"/paper/2607.14144"},"observation_digest":"sha256:cbb446b529edeee90c6ef3bf1a35268511bec8312a08a210511ab52a4ce5e6f3","observation_id":"8b9a51b3-bdaf-4644-83ec-dc2a6ca9bd5e","resolution":{"observed_at":"2026-08-03T02:03:25.442040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.00658/citation-record","integrity":"/paper/2501.00658/integrity","json":"/paper/2501.00658/citation-record.json","paper":"/paper/2501.00658"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.01590","last_updated":"2024-03-31T14:31:14Z","snapshot_observed_at":"2026-07-06T17:38:51.653737Z","submitted_at":"2024-03-03T18:58:21Z","title":"The Hidden Attention of Mamba Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.01590","snapshot_observed_at":"2026-08-10T22:52:33.343979Z","title":"The hidden attention of mamba models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.343979Z"},"links":{"cited_paper":"/paper/2403.01590","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:9a1be29e64e0f5893d2486a963670ad0330ba81bcfd885f441b6e32c0d2a9184","observation_id":"db420ba5-9de6-4c93-839a-53348ab29c1a","resolution":{"observed_at":"2026-08-10T22:52:33.343979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14528","last_updated":"2025-04-09T22:43:46Z","snapshot_observed_at":"2026-08-11T15:10:43.506804Z","submitted_at":"2024-06-20T17:40:18Z","title":"DeciMamba: Exploring the Length Extrapolation Potential of Mamba","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14528","snapshot_observed_at":"2026-08-10T22:52:33.368101Z","title":"Decimamba: Exploring the length extrapolation potential of mamba","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.368101Z"},"links":{"cited_paper":"/paper/2406.14528","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:f837f84454949203b4eb9a1efcdf1c9abcce6160e87d5242e7fd84de5299b23d","observation_id":"e0d11af2-eaeb-46be-b550-ef666c06c5e9","resolution":{"observed_at":"2026-08-10T22:52:33.368101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:33.373579Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.373579Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:0b7aef01f061924fdaf4c31112ff5afc2460a0e7c13ed3cf3ac6642929c2728c","observation_id":"b3726c25-d3c1-4271-b87f-f66b23dcd435","resolution":{"observed_at":"2026-08-10T22:52:33.373579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1406.1078","last_updated":"2014-09-03T00:25:02Z","snapshot_observed_at":"2026-07-06T03:45:28.546418Z","submitted_at":"2014-06-03T17:47:08Z","title":"Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1406.1078","snapshot_observed_at":"2026-08-10T22:52:33.379010Z","title":"Learning phrase representations using rnn encoder-decoder for statistical machine translation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.379010Z"},"links":{"cited_paper":"/paper/1406.1078","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:ca5112cebbbd26b2ea455f3f1d7c40ef8ec3f4b15a159ad6636b8edb86538f43","observation_id":"0b53bfbc-b713-4d11-b675-4c8034e50fd4","resolution":{"observed_at":"2026-08-10T22:52:33.379010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.14794","last_updated":"2022-11-19T12:45:21Z","snapshot_observed_at":"2026-08-12T04:58:34.201421Z","submitted_at":"2020-09-30T17:09:09Z","title":"Rethinking Attention with Performers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.14794","snapshot_observed_at":"2026-08-10T22:52:33.391034Z","title":"Rethinking attention with performers","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.391034Z"},"links":{"cited_paper":"/paper/2009.14794","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:7d697f706dcd1659ba036c347e134b28fe94378628ced3e0fd77276f05e88caf","observation_id":"e648a296-7d88-4cd8-9573-f06a8ec462e6","resolution":{"observed_at":"2026-08-10T22:52:33.391034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1901.02860","last_updated":"2019-06-02T21:21:48Z","snapshot_observed_at":"2026-07-06T07:25:48.468658Z","submitted_at":"2019-01-09T18:28:19Z","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1901.02860","snapshot_observed_at":"2026-08-10T22:52:33.396665Z","title":"Transformer-xl: Attentive language models beyond a fixed-length context","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.396665Z"},"links":{"cited_paper":"/paper/1901.02860","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:e330eba70b2e4f1a2ee2d2a7d540289c817bc74648e20843a6be244730f08e22","observation_id":"ad2c096e-eb0b-4ada-97f1-f0301739c813","resolution":{"observed_at":"2026-08-10T22:52:33.396665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19427","last_updated":"2024-02-29T18:24:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-29T18:24:46Z","title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19427","snapshot_observed_at":"2026-08-10T22:52:33.407687Z","title":"Griffin: Mix- ing gated linear recurrences with local attention for efficient language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.407687Z"},"links":{"cited_paper":"/paper/2402.19427","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:1a2782a1556e8172ffc2ba374443fbe945b74bc0fd25c53ecd27703d59ac31e8","observation_id":"ff742bf8-2803-48b4-990b-dd784d4ac900","resolution":{"observed_at":"2026-08-10T22:52:33.407687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.264174Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding","venue":null,"work_id":"ba01d256-03fd-497e-b14c-21e6efc53400","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.413833Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:a3412640336a0c2457c5096cd1c07a5247e15bde4ff46ff66b9fcf55d0c60669","observation_id":"46ff30e7-9068-4de8-a393-2d3f1182e06a","resolution":{"observed_at":"2026-08-10T22:52:35.269547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01201","last_updated":"2024-11-28T07:10:33Z","snapshot_observed_at":"2026-08-11T06:46:35.038684Z","submitted_at":"2024-10-02T03:06:49Z","title":"Were RNNs All We Needed?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01201","snapshot_observed_at":"2026-08-10T22:52:33.425335Z","title":"Were rnns all we needed? arXiv preprint arXiv:2410.01201,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.425335Z"},"links":{"cited_paper":"/paper/2410.01201","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:7bf5ede9a701ebc598fc01599346435d5aa37f1d1abc377fb23b9016cc456136","observation_id":"65edb53b-c766-4914-a89f-1d00d1585662","resolution":{"observed_at":"2026-08-10T22:52:33.425335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.14052","last_updated":"2023-04-29T03:18:40Z","snapshot_observed_at":"2026-08-06T18:01:24.576458Z","submitted_at":"2022-12-28T17:56:03Z","title":"Hungry Hungry Hippos: Towards Language Modeling with State Space Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.14052","snapshot_observed_at":"2026-08-10T22:52:33.430458Z","title":"Hungry hungry hippos: Towards language modeling with state space models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.430458Z"},"links":{"cited_paper":"/paper/2212.14052","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:be672c3a86050f222afe6a83f284fa9f7ce24e3b7de57e8998c9ed77088748ae","observation_id":"8701a239-5a12-4284-9d32-4c65459f1f28","resolution":{"observed_at":"2026-08-10T22:52:33.430458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.00396","last_updated":"2022-08-05T17:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-10-31T03:32:18Z","title":"Efficiently Modeling Long Sequences with Structured State Spaces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.00396","snapshot_observed_at":"2026-08-10T22:52:33.445941Z","title":"Efficiently modeling long sequences with structured state spaces","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.445941Z"},"links":{"cited_paper":"/paper/2111.00396","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:35edb9437f0ce3e6b871115cc99556c2e4ffc57da4c1cc0a96eeb4e5b2aca0e6","observation_id":"f44c4f39-53a2-4934-a11d-6ed4523c579c","resolution":{"observed_at":"2026-08-10T22:52:33.445941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.248380Z","title":"Long short-term memory","venue":null,"work_id":"c86351e6-0609-45ba-a891-bccce78e74b3","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.450952Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:b1a8cbc092d8633df3b855271ce90515fd0908f4325a4e3a08c33f83024179d5","observation_id":"52e4a487-a12e-4691-b1b3-24fed19f0800","resolution":{"observed_at":"2026-08-10T22:52:35.253392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01032","last_updated":"2024-06-03T22:22:15Z","snapshot_observed_at":"2026-08-09T22:32:41.851014Z","submitted_at":"2024-02-01T21:44:11Z","title":"Repeat After Me: Transformers are Better than State Space Models at Copying","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01032","snapshot_observed_at":"2026-08-10T22:52:33.461185Z","title":"Repeat after me: Trans- formers are better than state space models at copying","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.461185Z"},"links":{"cited_paper":"/paper/2402.01032","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:c70aeb3420fb9f9592eb26a4c7324470739df3e8658328a4ac8e1e76b2ed7dc3","observation_id":"196423ba-964a-463b-87b4-7a607dd2f3bc","resolution":{"observed_at":"2026-08-10T22:52:33.461185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-10T22:52:33.466271Z","title":"Mistral 7b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.466271Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:6c2be9bb5c0ffb66a9b2ebfffd68203fa21e903bf91945b72187f25c79d2aef9","observation_id":"7a60ab9c-7c21-4ff1-a378-ee88380691d1","resolution":{"observed_at":"2026-08-10T22:52:33.466271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-10T22:52:33.471740Z","title":"Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.471740Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:411b7e85b77e341100b027587183f1f562e4f862c8b725ccf8f3fca09f243488","observation_id":"4d0cfb62-0c7d-4ea3-8edd-387a9418404a","resolution":{"observed_at":"2026-08-10T22:52:33.471740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1609.02907","last_updated":"2017-02-22T09:55:36Z","snapshot_observed_at":"2026-07-06T05:10:16.862707Z","submitted_at":"2016-09-09T19:48:41Z","title":"Semi-Supervised Classification with Graph Convolutional Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.02907","snapshot_observed_at":"2026-08-10T22:52:33.477934Z","title":"Semi-supervised classification with graph convolutional networks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.477934Z"},"links":{"cited_paper":"/paper/1609.02907","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:4638bd229eb7060c7303e7f475561c6bd1d716f8c7d6199fea8cb4313da1a140","observation_id":"cd04fae6-c460-4fb2-9b81-ae140044975a","resolution":{"observed_at":"2026-08-10T22:52:33.477934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19887","last_updated":"2024-07-03T14:30:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-28T23:55:06Z","title":"Jamba: A Hybrid Transformer-Mamba Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19887","snapshot_observed_at":"2026-08-10T22:52:33.498531Z","title":"Jamba: A hybrid transformer- mamba language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.498531Z"},"links":{"cited_paper":"/paper/2403.19887","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:a8c5ca2e62ae0cb175e85136028354ca049e9318c9113ac93a3acb4f8a4fd644","observation_id":"981a46df-05ba-48ed-ba67-ae8ce609bb7c","resolution":{"observed_at":"2026-08-10T22:52:33.498531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14207","last_updated":"2024-10-02T14:32:59Z","snapshot_observed_at":"2026-08-04T14:58:56.675148Z","submitted_at":"2024-07-19T11:12:08Z","title":"Longhorn: State Space Models are Amortized Online Learners","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14207","snapshot_observed_at":"2026-08-10T22:52:33.504123Z","title":"Longhorn: State space models are amortized online learners","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.504123Z"},"links":{"cited_paper":"/paper/2407.14207","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:180cccb7db130629fa2f7842b696701773975fb88fe381f9f7f81d17ef540a0d","observation_id":"604fde47-106f-4e2e-8fe9-8f6ca2dc52de","resolution":{"observed_at":"2026-08-10T22:52:33.504123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.10655","last_updated":"2023-01-28T06:33:20Z","snapshot_observed_at":"2026-08-07T01:23:02.071264Z","submitted_at":"2022-09-21T20:52:17Z","title":"Mega: Moving Average Equipped Gated Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.10655","snapshot_observed_at":"2026-08-10T22:52:33.509637Z","title":"Mega: moving average equipped gated attention","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.509637Z"},"links":{"cited_paper":"/paper/2209.10655","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:869e3bb538d8a241c2c7841ab94de2855898863a558a4c815991cc4203a59553","observation_id":"5a25b695-3eb7-42c2-bead-ee61821c716f","resolution":{"observed_at":"2026-08-10T22:52:33.509637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08801","last_updated":"2024-04-16T07:27:58Z","snapshot_observed_at":"2026-08-10T22:37:12.590597Z","submitted_at":"2024-04-12T20:28:14Z","title":"Megalodon: Efficient LLM Pretraining and Inference with Unlimited Context Length","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08801","snapshot_observed_at":"2026-08-10T22:52:33.515259Z","title":"Megalodon: Efficient llm pretraining and inference with unlimited context length","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.515259Z"},"links":{"cited_paper":"/paper/2404.08801","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:63fdd5254494b29fe3dbf84e451966fbcc10bf562e29257bcab22e8674ce7bc9","observation_id":"6b830b8c-4773-4ce7-a0a6-2fdf0ffdabd9","resolution":{"observed_at":"2026-08-10T22:52:33.515259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08819","last_updated":"2025-03-05T23:00:57Z","snapshot_observed_at":"2026-07-06T17:59:38.202928Z","submitted_at":"2024-04-12T21:30:06Z","title":"The Illusion of State in State-Space Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08819","snapshot_observed_at":"2026-08-10T22:52:33.520485Z","title":"The illusion of state in state-space models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.520485Z"},"links":{"cited_paper":"/paper/2404.08819","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:30d366e77de9cd9365e8758ecd85acf0f216c252eaa9ef5faa2ecdcee2456ebe","observation_id":"f7c675f7-ccbc-472d-b38e-bf0e31e13161","resolution":{"observed_at":"2026-08-10T22:52:33.520485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-10T22:52:33.525899Z","title":"In-context learning and induction heads","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.525899Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:3c53a01b661a7cf10579a9f27c46f05c33fcfd9e5e9360d21a404399b81ce2cb","observation_id":"5d2451af-a931-47a5-b4f3-c10f01d81e91","resolution":{"observed_at":"2026-08-10T22:52:33.525899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.231845Z","title":"Graph neural networks exponentially lose expressive power for node classification","venue":null,"work_id":"0e6d8f2e-2965-4ee3-98eb-757e4a11869b","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.531018Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:60aff0e733400bd3f8c7013ef9cef5729b44eb0e69e8c840789b0d5b09be04d4","observation_id":"94bd2d67-7ceb-4d68-b90d-af485a0d876c","resolution":{"observed_at":"2026-08-10T22:52:35.237621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04248","last_updated":"2024-04-25T17:01:52Z","snapshot_observed_at":"2026-08-10T21:22:29.950743Z","submitted_at":"2024-02-06T18:56:35Z","title":"Can Mamba Learn How to Learn? A Comparative Study on In-Context Learning Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04248","snapshot_observed_at":"2026-08-10T22:52:33.536068Z","title":"Can mamba learn how to learn? a comparative study on in-context learning tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.536068Z"},"links":{"cited_paper":"/paper/2402.04248","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:b85817664cd551859a89e355e8b52005a7ddb1e5127269884a2300d280f1fce1","observation_id":"bb41ab24-0105-441a-8d79-be64415d5e43","resolution":{"observed_at":"2026-08-10T22:52:33.536068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13048","last_updated":"2023-12-11T03:58:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-22T13:57:41Z","title":"RWKV: Reinventing RNNs for the Transformer Era","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13048","snapshot_observed_at":"2026-08-10T22:52:33.541213Z","title":"Rwkv: Reinventing rnns for the transformer era","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.541213Z"},"links":{"cited_paper":"/paper/2305.13048","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:58bfd2e730d719232432fe95e51c705f778351ffdf7be1fe722d1fbd12993661","observation_id":"aa7d45c6-9e40-4381-abdd-c0db406053e1","resolution":{"observed_at":"2026-08-10T22:52:33.541213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05892","last_updated":"2024-09-26T22:39:08Z","snapshot_observed_at":"2026-08-10T10:37:29.996750Z","submitted_at":"2024-04-08T22:20:59Z","title":"Eagle and Finch: RWKV with Matrix-Valued States and Dynamic Recurrence","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05892","snapshot_observed_at":"2026-08-10T22:52:33.546904Z","title":"Eagle and finch: Rwkv with matrix-valued states and dynamic recurrence","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.546904Z"},"links":{"cited_paper":"/paper/2404.05892","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:f6b42224a24a1d48d9b7404dae259ee135e55618e60f5fd63a0232e2fd9dc469","observation_id":"f1fe9ae7-d4f5-4272-806a-a7f3dccc2fcd","resolution":{"observed_at":"2026-08-10T22:52:33.546904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09527","last_updated":"2022-11-17T13:43:20Z","snapshot_observed_at":"2026-07-06T14:19:47.424778Z","submitted_at":"2022-11-17T13:43:20Z","title":"Ignore Previous Prompt: Attack Techniques For Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09527","snapshot_observed_at":"2026-08-10T22:52:33.552154Z","title":"Ignore previous prompt: Attack techniques for language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.552154Z"},"links":{"cited_paper":"/paper/2211.09527","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:50ebc06d44d7d649a2d15d1f53924e9a8dab6f9748ef46dc249a6846e6ea650b","observation_id":"31794b19-e32f-462e-a7be-4115a891856a","resolution":{"observed_at":"2026-08-10T22:52:33.552154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17844","last_updated":"2024-08-19T17:26:18Z","snapshot_observed_at":"2026-08-09T14:35:48.289765Z","submitted_at":"2024-03-26T16:33:12Z","title":"Mechanistic Design and Scaling of Hybrid Architectures","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17844","snapshot_observed_at":"2026-08-10T22:52:33.557438Z","title":"Mechanistic design and scaling of hybrid architectures","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.557438Z"},"links":{"cited_paper":"/paper/2403.17844","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:b5c7d85bb936edd864a55607ba1f9067b7459efa906f8696f7a6c368ccda2090","observation_id":"3313a84a-12f0-41a4-9f60-523c149c119c","resolution":{"observed_at":"2026-08-10T22:52:33.557438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.12409","last_updated":"2022-04-22T18:20:48Z","snapshot_observed_at":"2026-08-07T11:26:24.970964Z","submitted_at":"2021-08-27T17:35:06Z","title":"Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.12409","snapshot_observed_at":"2026-08-10T22:52:33.563025Z","title":"Train short, test long: Attention with linear biases enables input length extrapolation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.563025Z"},"links":{"cited_paper":"/paper/2108.12409","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:355f08ee1c84de58d5d8c64cc91740b90be4aaa0ada2143580fd4ec2336b2287","observation_id":"8b7c8b1c-75c6-4691-824a-824f3bc3daa1","resolution":{"observed_at":"2026-08-10T22:52:33.563025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07904","last_updated":"2024-08-19T17:16:55Z","snapshot_observed_at":"2026-08-09T13:09:57.457279Z","submitted_at":"2024-04-11T16:43:03Z","title":"HGRN2: Gated Linear RNNs with State Expansion","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07904","snapshot_observed_at":"2026-08-10T22:52:33.568763Z","title":"Hgrn2: Gated linear rnns with state expansion","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.568763Z"},"links":{"cited_paper":"/paper/2404.07904","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:e5fcf0af0d8e21c395c491c75d78585f91936a748be8279105fac21497c35f8a","observation_id":"bb9dc605-5b32-4362-95c2-dfc742b0af19","resolution":{"observed_at":"2026-08-10T22:52:33.568763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.08625","last_updated":"2022-02-17T12:20:52Z","snapshot_observed_at":"2026-07-31T09:00:36.259346Z","submitted_at":"2022-02-17T12:20:52Z","title":"Revisiting Over-smoothing in BERT from the Perspective of Graph","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.08625","snapshot_observed_at":"2026-08-10T22:52:33.574082Z","title":"Revisiting over-smoothing in bert from the perspective of graph","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.574082Z"},"links":{"cited_paper":"/paper/2202.08625","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:9aecc6a3f8b87390456ef481f11de1645af92698f74e1314b7229bd4cdbde2f5","observation_id":"a2869e78-4e65-4c9a-b094-5fb5bd046626","resolution":{"observed_at":"2026-08-10T22:52:33.574082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04620","last_updated":"2025-08-31T18:32:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-05T16:23:20Z","title":"Learning to (Learn at Test Time): RNNs with Expressive Hidden States","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04620","snapshot_observed_at":"2026-08-10T22:52:33.579902Z","title":"Learning to (learn at test time): Rnns with expressive hidden states","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.579902Z"},"links":{"cited_paper":"/paper/2407.04620","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:84cbb939021e6eeb3b7126d5d975770f2ba2cf98b64ee8721d6e364f262c470c","observation_id":"d590677e-9b9b-4c03-97ec-cee95a138271","resolution":{"observed_at":"2026-08-10T22:52:33.579902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10554","last_updated":"2022-12-20T18:56:20Z","snapshot_observed_at":"2026-08-09T00:51:36.452647Z","submitted_at":"2022-12-20T18:56:20Z","title":"A Length-Extrapolatable Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10554","snapshot_observed_at":"2026-08-10T22:52:33.584955Z","title":"A length-extrapolatable transformer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.584955Z"},"links":{"cited_paper":"/paper/2212.10554","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:635e14cb139bc5a7c159ec69886016cf64dd2dffc432dec328205b253bf024d3","observation_id":"f119f4a3-66a9-40a9-afc9-cff89dbd82a2","resolution":{"observed_at":"2026-08-10T22:52:33.584955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08621","last_updated":"2023-08-09T08:53:08Z","snapshot_observed_at":"2026-08-02T13:21:32.251959Z","submitted_at":"2023-07-17T16:40:01Z","title":"Retentive Network: A Successor to Transformer for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08621","snapshot_observed_at":"2026-08-10T22:52:33.589977Z","title":"Retentive network: A successor to transformer for large language models.arXiv preprint arXiv:2307.08621,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.589977Z"},"links":{"cited_paper":"/paper/2307.08621","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:68be2e998dc44ccc077132cc33e9026e5387341cfbf5ba9f58ae3b9cb5c14cc3","observation_id":"38c5dc08-a5b4-4e9f-8c95-2a9bc85b24e6","resolution":{"observed_at":"2026-08-10T22:52:33.589977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2011.04006","last_updated":"2020-11-08T15:53:56Z","snapshot_observed_at":"2026-08-09T15:12:21.086848Z","submitted_at":"2020-11-08T15:53:56Z","title":"Long Range Arena: A Benchmark for Efficient Transformers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2011.04006","snapshot_observed_at":"2026-08-10T22:52:33.595036Z","title":"Long range arena: A benchmark for efficient transformers","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.595036Z"},"links":{"cited_paper":"/paper/2011.04006","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:ca8944dfc09834c19491023e6afdbdae3d9e7398ae8be47a9c7e9c439d4c3042","observation_id":"7ee84a7d-addb-4452-8dc3-4ffec2706397","resolution":{"observed_at":"2026-08-10T22:52:33.595036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.14522","last_updated":"2022-11-12T16:11:19Z","snapshot_observed_at":"2026-08-10T10:56:00.667216Z","submitted_at":"2021-11-29T13:27:56Z","title":"Understanding over-squashing and bottlenecks on graphs via curvature","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.14522","snapshot_observed_at":"2026-08-10T22:52:33.600245Z","title":"Understanding over-squashing and bottlenecks on graphs via curvature","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.600245Z"},"links":{"cited_paper":"/paper/2111.14522","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:8916e6df3b1ec2f9a3341eae3ea1b686c6f1856f1f8062e8a3fb287ce722f437","observation_id":"7232c5e4-5021-4556-a012-8dae7549c9aa","resolution":{"observed_at":"2026-08-10T22:52:33.600245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.11775","last_updated":"2019-11-11T21:51:11Z","snapshot_observed_at":"2026-08-07T12:23:46.977643Z","submitted_at":"2019-08-30T15:05:02Z","title":"Transformer Dissection: A Unified Understanding of Transformer's Attention via the Lens of Kernel","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.11775","snapshot_observed_at":"2026-08-10T22:52:33.605463Z","title":"Transformer dissection: a unified understanding of transformer’s attention via the lens of kernel","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.605463Z"},"links":{"cited_paper":"/paper/1908.11775","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:8ae298c19a68d32f7bf8e8e78baa99189ec7edc6aa909b057122e9492ce79677","observation_id":"820c490b-6f6a-4707-8891-06174adc6209","resolution":{"observed_at":"2026-08-10T22:52:33.605463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1804.04849","last_updated":"2018-09-13T10:55:56Z","snapshot_observed_at":"2026-07-06T06:33:15.714029Z","submitted_at":"2018-04-13T09:18:17Z","title":"The unreasonable effectiveness of the forget gate","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.04849","snapshot_observed_at":"2026-08-10T22:52:33.610896Z","title":"The unreasonable effectiveness of the forget gate","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.610896Z"},"links":{"cited_paper":"/paper/1804.04849","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:1847f834cc850d749844f57755e1e9caee997dad5b2f92abf61187d18bce6144","observation_id":"c0f19b57-13a7-4776-a1a8-7d0ea1f40e1c","resolution":{"observed_at":"2026-08-10T22:52:33.610896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07887","last_updated":"2024-06-12T05:25:15Z","snapshot_observed_at":"2026-07-06T18:29:21.709395Z","submitted_at":"2024-06-12T05:25:15Z","title":"An Empirical Study of Mamba-based Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07887","snapshot_observed_at":"2026-08-10T22:52:33.615941Z","title":"An empirical study of mamba- based language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.615941Z"},"links":{"cited_paper":"/paper/2406.07887","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:0227edf5230984f1205792b78b024a4059b5ac1bcfcebb0a72546afdc9f7abc2","observation_id":"3eac356a-996c-4591-981c-374ea6ebbfaa","resolution":{"observed_at":"2026-08-10T22:52:33.615941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.05962","last_updated":"2022-03-09T23:55:24Z","snapshot_observed_at":"2026-07-06T12:46:48.184877Z","submitted_at":"2022-03-09T23:55:24Z","title":"Anti-Oversmoothing in Deep Vision Transformers via the Fourier Domain Analysis: From Theory to Practice","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.05962","snapshot_observed_at":"2026-08-10T22:52:33.621004Z","title":"Anti-oversmoothing in deep vision transformers via the fourier domain analysis: From theory to practice","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.621004Z"},"links":{"cited_paper":"/paper/2203.05962","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:7b52d46af1fcb45bd10114ab7a603a0a4666e8676996c530ffd631b3f5a261aa","observation_id":"9fae8463-f14f-42d4-8f5a-1a071cc6734e","resolution":{"observed_at":"2026-08-10T22:52:33.621004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10701","last_updated":"2023-03-01T04:22:54Z","snapshot_observed_at":"2026-07-06T14:33:16.914984Z","submitted_at":"2022-12-21T00:33:59Z","title":"A Non-Asymptotic Analysis of Oversmoothing in Graph Neural Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10701","snapshot_observed_at":"2026-08-10T22:52:33.626180Z","title":"A non-asymptotic analysis of oversmoothing in graph neural networks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.626180Z"},"links":{"cited_paper":"/paper/2212.10701","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:672a656ec9344ccc3a3555890dd36e8d118ac90c13096a5446fb3fc0a81b92cc","observation_id":"ff99a6c3-f4c5-4123-957e-d1daa19e3488","resolution":{"observed_at":"2026-08-10T22:52:33.626180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.18781","last_updated":"2024-11-01T01:45:27Z","snapshot_observed_at":"2026-08-11T15:12:58.393972Z","submitted_at":"2024-05-29T05:41:28Z","title":"On the Role of Attention Masks and LayerNorm in Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.18781","snapshot_observed_at":"2026-08-10T22:52:33.631796Z","title":"On the role of attention masks and layernorm in transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.631796Z"},"links":{"cited_paper":"/paper/2405.18781","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:1b15243bee0b690a12ef9be5dbdd355f596db9f459b56dc8dabe4127193111aa","observation_id":"ab356f8d-85e6-4549-bff7-1b23830adf12","resolution":{"observed_at":"2026-08-10T22:52:33.631796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06635","last_updated":"2024-08-27T01:27:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-11T18:51:59Z","title":"Gated Linear Attention Transformers with Hardware-Efficient Training","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06635","snapshot_observed_at":"2026-08-10T22:52:33.636905Z","title":"Gated linear attention transformers with hardware-efficient training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.636905Z"},"links":{"cited_paper":"/paper/2312.06635","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:f174bdcb8870d2b9765e7c0211b9c34bb988bfe7225490f6fb396f79ede7c22f","observation_id":"4aa22575-10a5-4c71-99b3-af692f1008c1","resolution":{"observed_at":"2026-08-10T22:52:33.636905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06484","last_updated":"2025-01-15T10:41:40Z","snapshot_observed_at":"2026-08-06T05:53:13.494942Z","submitted_at":"2024-06-10T17:24:42Z","title":"Parallelizing Linear Transformers with the Delta Rule over Sequence Length","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06484","snapshot_observed_at":"2026-08-10T22:52:33.642183Z","title":"Parallelizing linear transformers with the delta rule over sequence length","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.642183Z"},"links":{"cited_paper":"/paper/2406.06484","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:ab68732f05fd20ebe311064bc30007179f019f1ceb8b103a448e1dad5d69f0b5","observation_id":"4ba87f36-9fb3-4c98-b0dc-8822888cfb0e","resolution":{"observed_at":"2026-08-10T22:52:33.642183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04347","last_updated":"2024-02-06T19:31:26Z","snapshot_observed_at":"2026-08-07T03:58:02.059388Z","submitted_at":"2024-02-06T19:31:26Z","title":"The Hedgehog & the Porcupine: Expressive Linear Attentions with Softmax Mimicry","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04347","snapshot_observed_at":"2026-08-10T22:52:33.648381Z","title":"The hedgehog & the porcupine: Expressive linear attentions with softmax mimicry","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.648381Z"},"links":{"cited_paper":"/paper/2402.04347","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:4c3f706813bbc1d75b8c79ac3442308ab311d76241819851297b61df81bb839a","observation_id":"0ca922e9-e7f9-4072-98f4-df3502f21d96","resolution":{"observed_at":"2026-08-10T22:52:33.648381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-08-12T09:06:50.363435Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-10T22:52:33.653929Z","title":"Universal and transferable adversarial attacks on aligned language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.653929Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:4537eea4c265cdcc7827569cf2f1a8dcfbf14a4f6b595c7af2684bb67d847253","observation_id":"7bea6beb-642d-4d91-8fa1-7f41c625fcc8","resolution":{"observed_at":"2026-08-10T22:52:33.653929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.215868Z","title":"Arora et al","venue":null,"work_id":"0ef832b7-bb78-4b76-836a-4e3c1c16d15a","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.659298Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:51a4f15de2aaff5f4953408928d8b4b172c32a99c46939f00bef5469c43f8d70","observation_id":"efe2d6c0-284e-4b5e-8d57-d9a70d40839b","resolution":{"observed_at":"2026-08-10T22:52:35.221479Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.199751Z","title":"Linear Attention","venue":null,"work_id":"cf7748ce-46fa-4e97-828a-75731976ea3d","year":2019},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.664625Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:4d0984bcacb9f55df60b9f73dd67bfafd8d67607a9cbd8629ab1240f617c8c38","observation_id":"cec37677-5009-4c35-9a1f-1d003b909a9d","resolution":{"observed_at":"2026-08-10T22:52:35.204856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.182570Z","title":"Each layer of RetNet consists of a key, 16 Published as a conference paper at ICLR 2025 query, and value transformation, akin to linear attention","venue":null,"work_id":"b552fe2d-0f8e-4492-86ee-5fd10a679878","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.669461Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:747d86a88a99b5ebcfe18273ac4ab944b8fe7e2cb302fa0143e07751b9bcf27f","observation_id":"988a1212-ba25-431d-ae6f-15c8d3853831","resolution":{"observed_at":"2026-08-10T22:52:35.188224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.165753Z","title":"Similar to Mamba (Gu & Dao, 2023), RetNet shares bt and ct across channels while assigning distinct ∆t for each channel when handling multi-channel inputs","venue":null,"work_id":"01418856-3d20-49d0-a660-4347f5b165f1","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.674872Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:6052e04b3699fc3bfa8f538a27c912f5135850e2bc35fd8b8caafbadc8e33b09","observation_id":"ca99d2e5-b3ea-427a-8027-19f3f5abb175","resolution":{"observed_at":"2026-08-10T22:52:35.171024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.148544Z","title":"Its computational mechanism can be encompassed by our formulation in Eq","venue":null,"work_id":"40fe464b-03ba-47ea-8d1c-536462a63034","year":2018},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.679907Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:425fd2edbe4d599f42f38fadbdebb7604dbcefffc38d5386768b0942ba18a42e","observation_id":"8be4f9ad-4423-4eff-b9f8-e7f7cebadf1f","resolution":{"observed_at":"2026-08-10T22:52:35.153508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.132974Z","title":null,"venue":null,"work_id":"8b050e43-9092-47b5-bcfd-3a5f1c0af199","year":2024},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.684676Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:de88ac3c901e2907ebb2911f5f4fe6ab81f2482cefa1c0981643ee970a90ab56","observation_id":"b2504547-851c-4e66-9a8b-e242a8c6d643","resolution":{"observed_at":"2026-08-10T22:52:35.137753Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.114792Z","title":null,"venue":null,"work_id":"5d4bccea-0c82-46fd-9fef-0e8ee9ef6121","year":2024},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.689634Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:1fdd1c45b9275e9f12fc4abd2a514b95ab6d281d419eea721ab41802210b3ff2","observation_id":"9c86dd9e-3253-4af5-a355-f7c35df7859b","resolution":{"observed_at":"2026-08-10T22:52:35.120957Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.091450Z","title":"In particular, the dimension of ht in Griffin is equal to the dimension of xt","venue":null,"work_id":"a0ab5b93-e9a0-41b5-b638-4f4577951875","year":2020},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.694930Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:3a7606bdc922609582f85e1b3198c470a22b7e32cdd0342ef679df0136d93bef","observation_id":"b58c48aa-d56c-4ae0-a6a7-eb3cd43f4f80","resolution":{"observed_at":"2026-08-10T22:52:35.099255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.074191Z","title":"This design has quickly become a standard backbone for various SSMs (Gu & Dao, 2023; Beck et al., 2024)","venue":null,"work_id":"084fe9ac-835c-4eaf-82ed-a4baa74e54c6","year":2021},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.700455Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:d24abaed84a1a320fd84ee4890eccacbca320381d8fd14b32aea5fc2c17419c2","observation_id":"2c4ae5a1-9287-43b2-be62-a1da2360be1e","resolution":{"observed_at":"2026-08-10T22:52:35.079906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.056098Z","title":"attention","venue":null,"work_id":"5d55c2d9-8f2d-450e-b3e5-00a724a26f2a","year":2024},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.708740Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:11ccec5e30f1ded500e205ff9adc7a27ebbcd60549f6cfb196d6e0c7240f102b","observation_id":"0ad87dfd-bb5e-4198-97de-4fbe5e2d05d6","resolution":{"observed_at":"2026-08-10T22:52:35.062080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.038944Z","title":"We consider ϵ >0 small enough, thus, it is sufficient to consider the scenario when |ω| > Amax ≜ maxn∈[N ] |An,n|","venue":null,"work_id":"18d261bf-efce-4989-9023-0c3d612b6d20","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.714780Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:9120a619c6dbe929efce29f22c00815309351a65de8cdbfc5666ab0ef3a4f17c","observation_id":"f362ba20-f132-47cb-9635-ea2101b7fd0b","resolution":{"observed_at":"2026-08-10T22:52:35.044358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:35.022517Z","title":"Furthermore, let q = 1 − p","venue":null,"work_id":"eca91925-017e-4c50-8f27-065a41bf66f4","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.719862Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:b6fd613e2b2a8ff15a1afe50b947fa0d110aa16c341c9c6db1920bf8591736c1","observation_id":"60b5fc55-623d-445f-9c91-633d71208f2a","resolution":{"observed_at":"2026-08-10T22:52:35.027899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.886359Z","title":"less than","venue":null,"work_id":"313a8504-da8f-4b25-a203-f4ef16335ce5","year":2024},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.724979Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:f2360d547fab15565da0cce8e3dc987211b293195339963721080bd4a3cb0a6f","observation_id":"0ff8a946-7cf4-42f8-86cc-c3260a97e1d5","resolution":{"observed_at":"2026-08-10T22:52:35.010124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.870055Z","title":"Needle in a Haystack","venue":null,"work_id":"2ea70ce3-e2e4-4972-a128-f24a77b77983","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.731116Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:7697ce5ad9a0e5eaf2ceec52edba9138121c41a298a173f574acf21d0d2e8f79","observation_id":"e199a764-7053-462f-a8e5-97568a4d0196","resolution":{"observed_at":"2026-08-10T22:52:34.875178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.853460Z","title":"In SSMs, the class token must be positioned last to aggregate features from the entire sequence","venue":null,"work_id":"8791413e-b7d4-423a-959a-afd5cfb2bfa1","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.736023Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:414ba6946925fc889258a83e40393eb61695c17a92e8d4cdc9a84b5003fd9bc2","observation_id":"92ff088c-6436-4679-92dd-0e678b6714e9","resolution":{"observed_at":"2026-08-10T22:52:34.858842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.834835Z","title":"In addition, our image classification setup differs from Tay et al","venue":null,"work_id":"6144d622-34e5-486c-8212-e0653cf57b48","year":2020},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.741482Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:073fa0478bfe40a4b6b107b6eda90a671dd4674761faa44dc45a4e73dd90fcf3","observation_id":"33ea289f-1c1b-45b4-92cf-b1dc038a563e","resolution":{"observed_at":"2026-08-10T22:52:34.840131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.817227Z","title":"The models and training pipelines are built on Arora et al","venue":null,"work_id":"4cd411fa-7591-4a9d-81e1-ed948040ff7e","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.747493Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:7b66ee5cca41b641ac1f973761832c7018a95c4d05df2b49ef8e79ede9174c9e","observation_id":"cef29079-19c9-46f9-b0d9-b9f933e1ea1d","resolution":{"observed_at":"2026-08-10T22:52:34.822951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.795819Z","title":"For each curve in Fig","venue":null,"work_id":"a17ff03a-2dfe-43f4-af86-c803bec1c2ab","year":2025},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.752839Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:63c246d1eee796fbbb331098953522b9dbb62d9f35e946de808843a679535f03","observation_id":"688f22a6-b7d7-45c6-a3ba-1edfb924e787","resolution":{"observed_at":"2026-08-10T22:52:34.803795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.775656Z","title":"The evaluation set is created by holding out a subset of 10M tokens from the training data","venue":null,"work_id":"e07732ad-0aa6-4800-9699-654e7a56e3a4","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.757757Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:eb2c7148d35aeb16d51d3980cbe22e505ca3d84734520502b95394824137d096","observation_id":"6bf046c4-9441-4621-a89b-01a5b3a1aa9e","resolution":{"observed_at":"2026-08-10T22:52:34.781393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.759670Z","title":"We test two block sizes {2048, 8192}","venue":null,"work_id":"0e77d0b9-49a1-4577-9e2c-6458f41c3c30","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.762879Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:da1c8286f74eb55375392aeffab8b9e94b359d01eb399620ecc8a9c04ddc39c2","observation_id":"dc07023b-8caf-43bc-ad5c-0828f5bcd519","resolution":{"observed_at":"2026-08-10T22:52:34.764751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.742517Z","title":"0 A −1000 # , At ≈","venue":null,"work_id":"607b766f-ec1c-469f-97a1-0111b8e52250","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.767881Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:5a386d6a390c41575682e06f4aedc366875ec3ed125d44156e5569fc819f37a1","observation_id":"682063cf-cf7c-4876-a815-9a389b0b523b","resolution":{"observed_at":"2026-08-10T22:52:34.748149Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:52:34.723246Z","title":"We consider 1-polarization mitigates locality most significantly, while deepening architecture only relieves recency mildly but deteriorates over-smoothing","venue":null,"work_id":"421d07f7-46b7-4aaf-bf04-3358409250e8","year":2023},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.772825Z"},"links":{"citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:249f704a6079c4cdf6093e1080425c68048511a01cbaf69e732b51aa65dc2002","observation_id":"55b2cff7-9f66-465f-9b3f-8f1d9719e749","resolution":{"observed_at":"2026-08-10T22:52:34.730788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.15556","last_updated":"2022-03-29T13:38:03Z","snapshot_observed_at":"2026-08-09T19:52:33.533277Z","submitted_at":"2022-03-29T13:38:03Z","title":"Training Compute-Optimal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.15556","snapshot_observed_at":"2026-08-10T22:52:33.455866Z","title":"Training compute-optimal large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":1997,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.455866Z"},"links":{"cited_paper":"/paper/2203.15556","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:83e34b7094cfca66a7e731bf66edbc2c980049b8c5ae858e7d8a566f65ff5ee8","observation_id":"c812e534-2070-4e09-9dd2-d6c59daae234","resolution":{"observed_at":"2026-08-10T22:52:33.455866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11794","snapshot_observed_at":"2026-08-10T22:52:33.483637Z","title":"Datacomp-lm: In search of the next generation of training sets for language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":1998,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.483637Z"},"links":{"cited_paper":"/paper/2406.11794","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:a6dd66769d704617c7823b05bed38b44972d477c384fe3d0e88a1368afd1165c","observation_id":"dbbac38c-b5cb-42f7-9537-2af0e4d67ef8","resolution":{"observed_at":"2026-08-10T22:52:33.483637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1409.1259","last_updated":"2014-10-07T18:08:30Z","snapshot_observed_at":"2026-07-06T03:53:24.366023Z","submitted_at":"2014-09-03T21:03:41Z","title":"On the Properties of Neural Machine Translation: Encoder-Decoder Approaches","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.1259","snapshot_observed_at":"2026-08-10T22:52:33.384730Z","title":"On the properties of neural machine translation: Encoder-decoder approaches","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.384730Z"},"links":{"cited_paper":"/paper/1409.1259","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:8bcf51fbfb1876074325d50eeb31927dedde318ad402572c1152761afde207dd","observation_id":"88595fe3-5e42-42e3-9ba2-de9ce0d109b8","resolution":{"observed_at":"2026-08-10T22:52:33.384730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-10T22:52:33.440979Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.440979Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:5d0addc2697983a777dabf785e23bbdae961dac912759386a2c9307ca4dd37f3","observation_id":"2e6c85fc-6a4a-4d53-8292-32b5122f7358","resolution":{"observed_at":"2026-08-10T22:52:33.440979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09298","last_updated":"2022-10-17T17:53:29Z","snapshot_observed_at":"2026-08-10T14:44:20.342654Z","submitted_at":"2022-10-17T17:53:29Z","title":"What Makes Convolutional Models Great on Long Sequence Modeling?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.09298","snapshot_observed_at":"2026-08-10T22:52:33.489223Z","title":"What makes convolutional models great on long sequence modeling? arXiv preprint arXiv:2210.09298,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.489223Z"},"links":{"cited_paper":"/paper/2210.09298","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:00a7a964c9159b86a79f04aeac76da1a5171ee0c505e90085d2caca6ca55d32e","observation_id":"a86b3051-3238-40c5-b68c-7d4911d25845","resolution":{"observed_at":"2026-08-10T22:52:33.489223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21060","last_updated":"2024-05-31T17:50:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:50:01Z","title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21060","snapshot_observed_at":"2026-08-10T22:52:33.402204Z","title":"Transformers are ssms: Generalized models and efficient algorithms through structured state space duality","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.402204Z"},"links":{"cited_paper":"/paper/2405.21060","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:8b1ab9499a7eb4e65426f2a59b5f60074cc918dd80ba8243a132407f46670877","observation_id":"694c31f0-3289-4e9a-9462-624096c29bfd","resolution":{"observed_at":"2026-08-10T22:52:33.402204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04927","last_updated":"2023-12-08T09:44:25Z","snapshot_observed_at":"2026-08-05T07:11:50.808005Z","submitted_at":"2023-12-08T09:44:25Z","title":"Zoology: Measuring and Improving Recall in Efficient Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04927","snapshot_observed_at":"2026-08-10T22:52:33.356788Z","title":"Zoology: Measuring and improving recall in efficient language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.356788Z"},"links":{"cited_paper":"/paper/2312.04927","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:d7617639d70acf08e9c06fbdf5659396eb5c312e1957b480ae791609e0abb3bc","observation_id":"02387a9e-f650-4752-9771-413844168932","resolution":{"observed_at":"2026-08-10T22:52:33.356788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-10T22:52:33.419780Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.419780Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:ae4773e8ddb40b08c060a88474ba12f5155b5a63c903a58e8a490df06443dc28","observation_id":"30770a5f-7dcb-4956-a4ea-d4384b00656f","resolution":{"observed_at":"2026-08-10T22:52:33.419780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10794","last_updated":"2025-08-21T11:31:09Z","snapshot_observed_at":"2026-08-10T12:03:25.806428Z","submitted_at":"2023-12-17T19:06:29Z","title":"A mathematical perspective on Transformers","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10794","snapshot_observed_at":"2026-08-10T22:52:33.435966Z","title":"A mathematical perspec- tive on transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.435966Z"},"links":{"cited_paper":"/paper/2312.10794","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:f442208ff7a4f7848b10459221b7acaf19b800f0c4ac07d0a8e1572daf96ffb9","observation_id":"baced2b1-8fcc-4e09-adb2-d23b25c95fdb","resolution":{"observed_at":"2026-08-10T22:52:33.435966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04517","last_updated":"2024-12-06T15:42:07Z","snapshot_observed_at":"2026-08-07T15:21:45.322877Z","submitted_at":"2024-05-07T17:50:21Z","title":"xLSTM: Extended Long Short-Term Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04517","snapshot_observed_at":"2026-08-10T22:52:33.362373Z","title":"xlstm: Extended long short-term memory","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.362373Z"},"links":{"cited_paper":"/paper/2405.04517","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:9fcdec534cbea21d23cff3586d8331c517817b76a2edcb5e217ceedeb5391104","observation_id":"76ed0eb1-b777-4cee-a63a-b260004c07bf","resolution":{"observed_at":"2026-08-10T22:52:33.362373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.05205","last_updated":"2021-03-09T11:53:06Z","snapshot_observed_at":"2026-08-06T10:13:00.684351Z","submitted_at":"2020-06-09T12:04:50Z","title":"On the Bottleneck of Graph Neural Networks and its Practical Implications","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.05205","snapshot_observed_at":"2026-08-10T22:52:33.351307Z","title":"On the bottleneck of graph neural networks and its practical implications","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-10T22:52:33.351307Z"},"links":{"cited_paper":"/paper/2006.05205","citing_paper":"/paper/2501.00658"},"observation_digest":"sha256:87a9e6675cf11c52f23080e5989c11d7df8847d0c13dae9eb84c4f452b611a7a","observation_id":"64837234-4575-4e0b-b303-0bafb2f189b9","resolution":{"observed_at":"2026-08-10T22:52:33.351307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.00658","last_updated":"2025-03-11T03:58:57Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-12T06:36:12.188779Z","submitted_at":"2024-12-31T22:06:39Z","title":"Understanding and Mitigating Bottlenecks of State Space Models through the Lens of Recency and Over-smoothing"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":56,"verified_exact":0,"verified_fuzzy":23},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 4 inbound Pith citation observations for arXiv:2501.00658."}