{"as_of":"2026-08-23T12:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c61877cdec31c18c50df0dbe5eca42f7f2cee7828bdf40f4a858a0ef5a93b683","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T16:23:16.094743Z","state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.02032/citation-record","integrity":"/paper/2608.02032/integrity","json":"/paper/2608.02032/citation-record.json","paper":"/paper/2608.02032"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.17685","last_updated":"2024-11-26T18:52:06Z","snapshot_observed_at":"2026-08-18T06:32:27.100646Z","submitted_at":"2024-11-26T18:52:06Z","title":"Attamba: Attending To Multi-Token States","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17685","snapshot_observed_at":"2026-08-04T16:23:12.468968Z","title":"Attamba: Attending to multi-token states.arXiv preprint arXiv:2411.17685,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:12.468968Z"},"links":{"cited_paper":"/paper/2411.17685","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:421da64b68b01c8a11df46973739e880259b9170f9369d12abf0dc9c115f77a9","observation_id":"8539423f-5bdd-4887-909c-806ed7a90f7f","resolution":{"observed_at":"2026-08-04T16:23:12.468968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-08-14T19:36:07.505691Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-04T16:23:12.884752Z","title":"Think you have solved question answering? try ARC, the AI2 reasoning chal- lenge.arXiv preprint arXiv:1803.05457,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:12.884752Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:22590c76177bdd81b48ff143d117abc347911f18fa9ad6ebece7dd83c57ee070","observation_id":"29191822-520d-40a5-8d09-d9625091e65f","resolution":{"observed_at":"2026-08-04T16:23:12.884752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21060","last_updated":"2024-05-31T17:50:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:50:01Z","title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21060","snapshot_observed_at":"2026-08-04T16:23:13.174744Z","title":"Transformers are SSMs: Generalized models and efficient algorithms through structured state space duality.arXiv preprint arXiv:2405.21060,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:13.174744Z"},"links":{"cited_paper":"/paper/2405.21060","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:ece1661f989e5d0138b6a9afab7c2665f56d273208730eb9b8665072e495ab2b","observation_id":"d42f4839-a8b2-46af-a012-08198c9eaa30","resolution":{"observed_at":"2026-08-04T16:23:13.174744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16712","last_updated":"2024-05-26T22:23:02Z","snapshot_observed_at":"2026-08-18T10:18:01.875999Z","submitted_at":"2024-05-26T22:23:02Z","title":"Zamba: A Compact 7B SSM Hybrid Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16712","snapshot_observed_at":"2026-08-04T16:23:13.534743Z","title":"Zamba: A compact 7B SSM hybrid model.arXiv preprint arXiv:2405.16712,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:13.534743Z"},"links":{"cited_paper":"/paper/2405.16712","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:336f0c2e01440a1b570199b0bb430009f1ed08a093dc860077b6b5679970300c","observation_id":"0dc0081a-e71a-46bd-9e91-054b53168889","resolution":{"observed_at":"2026-08-04T16:23:13.534743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-08-17T20:47:46.242385Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-04T16:23:13.654907Z","title":"Mamba: Linear-time sequence modeling with selective state spaces.arXiv preprint arXiv:2312.00752,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:13.654907Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:c421ad986104e76b22b4d44524e7c75898a9d2f22c2cc946930f110d931d2bba","observation_id":"41529fa1-3355-4e8e-8964-13217bb32efb","resolution":{"observed_at":"2026-08-04T16:23:13.654907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01032","last_updated":"2024-06-03T22:22:15Z","snapshot_observed_at":"2026-08-21T15:16:48.017050Z","submitted_at":"2024-02-01T21:44:11Z","title":"Repeat After Me: Transformers are Better than State Space Models at Copying","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01032","snapshot_observed_at":"2026-08-04T16:23:14.385860Z","title":"Repeat after me: Trans- formers are better than state space models at copying.arXiv preprint arXiv:2402.01032,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:14.385860Z"},"links":{"cited_paper":"/paper/2402.01032","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:3cb830c85d43c94162107d43489839a8b1c84702876a8803456627b7868fab21","observation_id":"74de6bad-a1d0-4c22-af69-25568eff1569","resolution":{"observed_at":"2026-08-04T16:23:14.385860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19887","last_updated":"2024-07-03T14:30:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-28T23:55:06Z","title":"Jamba: A Hybrid Transformer-Mamba Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19887","snapshot_observed_at":"2026-08-04T16:23:14.806287Z","title":"Jamba: A hybrid transformer- Mamba language model.arXiv preprint arXiv:2403.19887,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:14.806287Z"},"links":{"cited_paper":"/paper/2403.19887","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:2889bf340f05c1ea171de42b185aa201ca2abf27d1f95edd121f43f69cef7b5d","observation_id":"4a83bfa8-a121-4b4c-8a1a-bab5ebd7fcad","resolution":{"observed_at":"2026-08-04T16:23:14.806287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04434","last_updated":"2024-06-19T06:04:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-07T15:56:43Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04434","snapshot_observed_at":"2026-08-04T16:23:15.004747Z","title":"DeepSeek-V2: A strong, economical, and efficient mixture- of-experts language model.arXiv preprint arXiv:2405.04434, 2024a","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:15.004747Z"},"links":{"cited_paper":"/paper/2405.04434","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:8403a4ce6c37b0b8df67013c88389c59560fff4f6161bcab91adcf465282830e","observation_id":"6cc8213f-f69f-4e40-9a7e-30d14b9e1961","resolution":{"observed_at":"2026-08-04T16:23:15.004747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:23:15.545025Z","title":"Samba: Simple hybrid state space models for efficient unlimited context language modeling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:15.545025Z"},"links":{"citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:9820a2c8b3d96e4e2900357af63ee61b0f5b7c4cbc5bfc3493a38271896431c9","observation_id":"68e105c5-4834-49ea-ba4e-94abf07cf497","resolution":{"observed_at":"2026-08-04T16:23:15.545025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08621","last_updated":"2023-08-09T08:53:08Z","snapshot_observed_at":"2026-08-18T21:18:20.077927Z","submitted_at":"2023-07-17T16:40:01Z","title":"Retentive Network: A Successor to Transformer for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08621","snapshot_observed_at":"2026-08-04T16:23:15.815056Z","title":"Retentive network: A successor to transformer for large language models.arXiv preprint arXiv:2307.08621,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:15.815056Z"},"links":{"cited_paper":"/paper/2307.08621","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:4d9dc2823fac1229c6f3e20afe22b9130936cfa55044fb5a54675cd1f3f7935e","observation_id":"bc972461-5a1d-46a1-9019-c3053e32186c","resolution":{"observed_at":"2026-08-04T16:23:15.815056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-04T16:23:15.954841Z","title":"LLaMA: Open and efficient foundation language models.arXiv preprint arXiv:2302.13971,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:15.954841Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:f11ee5a65c0e19d4f5f8f08aca79f54a1dc17ddafa871c8b039689ca41ee4357","observation_id":"e1cfaf4f-82ef-4323-bf9f-613962ef5469","resolution":{"observed_at":"2026-08-04T16:23:15.954841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00027","last_updated":"2020-12-31T19:00:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-12-31T19:00:10Z","title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00027","snapshot_observed_at":"2026-08-04T16:23:13.365450Z","title":"The Pile: An 800GB dataset of diverse text for language modeling.arXiv preprint arXiv:2101.00027,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":1990,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:13.365450Z"},"links":{"cited_paper":"/paper/2101.00027","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:48f457daaacefcad34d6ee75c9090d147647e320feb4541b2a14615b0bd2de2f","observation_id":"fe103f6b-2403-4d91-bb98-c93bdb2355fd","resolution":{"observed_at":"2026-08-04T16:23:13.365450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06654","last_updated":"2024-08-06T21:48:58Z","snapshot_observed_at":"2026-08-18T14:33:16.365411Z","submitted_at":"2024-04-09T23:41:27Z","title":"RULER: What's the Real Context Size of Your Long-Context Language Models?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.06654","snapshot_observed_at":"2026-08-04T16:23:14.121784Z","title":"RULER: What’s the real context size of your long-context language models?arXiv preprint arXiv:2404.06654,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":1997,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:14.121784Z"},"links":{"cited_paper":"/paper/2404.06654","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:bbd1d721805e99f8442bc65cb3d6ac275b4902c129acfff88270f626ff255e33","observation_id":"93d5b400-ebab-43e1-8af2-716264a7eea2","resolution":{"observed_at":"2026-08-04T16:23:14.121784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.22791","last_updated":"2026-05-21T17:44:57Z","snapshot_observed_at":"2026-08-17T19:58:31.549937Z","submitted_at":"2026-05-21T17:44:57Z","title":"Gated DeltaNet-2: Decoupling Erase and Write in Linear Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.22791","snapshot_observed_at":"2026-08-04T16:23:13.984846Z","title":"Gated DeltaNet-2: Decoupling erase and write in linear attention.arXiv preprint arXiv:2605.22791,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:13.984846Z"},"links":{"cited_paper":"/paper/2605.22791","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:9ebed565adaf8ed78e07cf7c29f6b073c4c21d434c2e05020b212a937b8a00c6","observation_id":"0644f03e-5047-4866-89af-e7bf7fa4b2cd","resolution":{"observed_at":"2026-08-04T16:23:13.984846Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:23:15.234741Z","title":"RWKV: Reinventing RNNs for the transformer era","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:15.234741Z"},"links":{"citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:7ed1770d505b9f01ff1ce55c0c84f2ed3884d7596ed67b6393ec472a0860b905","observation_id":"f96f38e5-48f7-4b0a-9bf8-08ff6006fe25","resolution":{"observed_at":"2026-08-04T16:23:15.234741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07887","last_updated":"2024-06-12T05:25:15Z","snapshot_observed_at":"2026-07-06T18:29:21.709395Z","submitted_at":"2024-06-12T05:25:15Z","title":"An Empirical Study of Mamba-based Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07887","snapshot_observed_at":"2026-08-04T16:23:16.094743Z","title":"An empirical study of Mamba- based language models.arXiv preprint arXiv:2406.07887,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:16.094743Z"},"links":{"cited_paper":"/paper/2406.07887","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:bfb703eccce8af254ea569b55174b28e0c65e37e700076f06eb86358de6ce415","observation_id":"2fd0833c-eee5-47a5-be9e-e8a9435eed6d","resolution":{"observed_at":"2026-08-04T16:23:16.094743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:23:12.984838Z","title":"FlashAttention-2: Faster attention with better parallelism and work partitioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:12.984838Z"},"links":{"citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:49ec3703d3dd605953d203f649c8ae0aa4b4836761282f0414bbdc6fe599a123","observation_id":"7a1bbf1a-3a5d-4124-9580-439ac0650f1e","resolution":{"observed_at":"2026-08-04T16:23:12.984838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:23:14.614811Z","title":"Li, Berlin Chen, Caitlin Wang, Aviv Bick, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:14.614811Z"},"links":{"citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:3ef1365687ae58aa50514ca9d1ce91f8077352e09a242ef707c5ce0d98635fde","observation_id":"a92629ae-e9ac-434b-98a6-be7c136bffa2","resolution":{"observed_at":"2026-08-04T16:23:14.614811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:23:12.744337Z","title":"Language models are few-shot learners.Advances in neural information processing systems, 33:1877–1901,","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:12.744337Z"},"links":{"citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:d4c9e35de9a191d0ae8000ca8fe5afa19aeb5cddc44b62c7e2775228ff67e615","observation_id":"52be26f5-8845-4aa7-8f6a-614cf4133422","resolution":{"observed_at":"2026-08-04T16:23:12.744337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2208.04933","last_updated":"2023-03-03T18:35:28Z","snapshot_observed_at":"2026-08-13T08:14:01.416880Z","submitted_at":"2022-08-09T17:57:43Z","title":"Simplified State Space Layers for Sequence Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.04933","snapshot_observed_at":"2026-08-04T16:23:15.687548Z","title":"Simplified state space layers for sequence modeling.arXiv preprint arXiv:2208.04933,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:15.687548Z"},"links":{"cited_paper":"/paper/2208.04933","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:8ff8c6628f4f82e34d74824a1c6e5d144d194d2ce3961df97184aa7f27a6fda5","observation_id":"462c56b4-1714-4ddc-abfd-d3c088b1e945","resolution":{"observed_at":"2026-08-04T16:23:15.687548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:23:13.234740Z","title":"DROP: A reading comprehension benchmark requiring discrete reasoning over paragraphs","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:13.234740Z"},"links":{"citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:457e7cd74ad6e3913d9e19403ab234fe2cf9cf73dd52c3864d4c65dc299d5104","observation_id":"a5ac6537-4fa8-4327-a38a-e30bb26ea72a","resolution":{"observed_at":"2026-08-04T16:23:13.234740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.00396","last_updated":"2022-08-05T17:54:38Z","snapshot_observed_at":"2026-08-14T01:02:41.198730Z","submitted_at":"2021-10-31T03:32:18Z","title":"Efficiently Modeling Long Sequences with Structured State Spaces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.00396","snapshot_observed_at":"2026-08-04T16:23:13.798657Z","title":"Efficiently modeling long sequences with structured state spaces.arXiv preprint arXiv:2111.00396,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:13.798657Z"},"links":{"cited_paper":"/paper/2111.00396","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:7e3f47f742d25b904b4a232d850174f8856a19c9e9cd3c01581480afaf0be0fe","observation_id":"8c6b10b9-dc50-4311-b5f6-2b2c1d1e50a1","resolution":{"observed_at":"2026-08-04T16:23:13.798657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18668","last_updated":"2025-03-07T18:57:52Z","snapshot_observed_at":"2026-08-16T14:14:23.399336Z","submitted_at":"2024-02-28T19:28:27Z","title":"Simple linear attention language models balance the recall-throughput tradeoff","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18668","snapshot_observed_at":"2026-08-04T16:23:12.574467Z","title":"Zoology: Measuring and improving recall in efficient language mod- els","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-04T16:23:12.574467Z"},"links":{"cited_paper":"/paper/2402.18668","citing_paper":"/paper/2608.02032"},"observation_digest":"sha256:a58b6801bc3978f5e82c60f32210e9b79d2d65749ada9dcaffcaf8ab7bf71fb0","observation_id":"965468b8-0054-4ffc-9657-b2b217569011","resolution":{"observed_at":"2026-08-04T16:23:12.574467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.02032","last_updated":"2026-08-03T10:30:27Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-18T14:03:26.386382Z","submitted_at":"2026-08-03T10:30:27Z","title":"DART: Decoded Attention over Recurrent States for Efficient Long-Context Sequence Modeling"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":23,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 0 inbound Pith citation observations for arXiv:2608.02032."}