{"as_of":"2026-08-07T11:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6ec132ac44f0498008e9f6a034c3107d4eb0b659adfce402f74ba30525502ea2","coverage":[{"denominator":17,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T07:44:18.354744Z","state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2510.24606/citation-record","integrity":"/paper/2510.24606/integrity","json":"/paper/2510.24606/citation-record.json","paper":"/paper/2510.24606"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2409.01495","last_updated":"2024-09-02T23:28:15Z","snapshot_observed_at":"2026-08-07T06:38:05.013043Z","submitted_at":"2024-09-02T23:28:15Z","title":"The Compressor-Retriever Architecture for Language Model OS","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01495","snapshot_observed_at":"2026-08-04T07:44:15.795227Z","title":"The compressor-retriever architecture for language model os.arXiv preprint arXiv:2409.01495,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:15.795227Z"},"links":{"cited_paper":"/paper/2409.01495","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:19615c8c6d3f4a861c2f9ae42c8ef92553187e7c46dfc8f19ed22c51aea289f2","observation_id":"a7ec3d41-8d54-4bb0-8924-a73383bd884b","resolution":{"observed_at":"2026-08-04T07:44:15.795227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-04T07:44:16.434996Z","title":"Gemma Team, Morgane Riviere, Shreya Pathak, Pier Giuseppe Sessa, Cassidy Hardin, Surya Bhupatiraju, Léonard Hussenot, Thomas Mesnard, Bobak Shahriari, Alexandre Ramé, et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:16.434996Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:9638dbf2d15e32649f6031e6cd7a308716e6ff77b32121820f0721ca58a3cf33","observation_id":"2e6cc209-8069-4c30-a792-be0f76ef7649","resolution":{"observed_at":"2026-08-04T07:44:16.434996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19786","last_updated":"2025-03-25T15:52:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-25T15:52:34Z","title":"Gemma 3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19786","snapshot_observed_at":"2026-08-04T07:44:16.564775Z","title":"Gemma 3 technical report.arXiv preprint arXiv:2503.19786,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:16.564775Z"},"links":{"cited_paper":"/paper/2503.19786","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:e07be20e823f22f23521f6366abd020685470ca77ae4f58365bc2cd92de7df51","observation_id":"1873a0f2-cba9-45ce-9fb3-a3311a189cf7","resolution":{"observed_at":"2026-08-04T07:44:16.564775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14482","last_updated":"2025-02-14T20:58:41Z","snapshot_observed_at":"2026-07-06T18:49:10.964379Z","submitted_at":"2024-07-19T17:35:47Z","title":"ChatQA 2: Bridging the Gap to Proprietary LLMs in Long Context and RAG Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14482","snapshot_observed_at":"2026-08-04T07:44:16.794770Z","title":"Chatqa 2: Bridging the gap to proprietary llms in long context and rag capabilities.arXiv preprint arXiv:2407.14482,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:16.794770Z"},"links":{"cited_paper":"/paper/2407.14482","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:600052fe5083079625ac642a9b49bc95be6c4f157e6741f106a233f4a47cc26f","observation_id":"4f1735dd-4f94-42a9-9304-23b6ea7908f2","resolution":{"observed_at":"2026-08-04T07:44:16.794770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02069","last_updated":"2025-05-15T17:18:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-04T07:51:30Z","title":"PyramidKV: Dynamic KV Cache Compression based on Pyramidal Information Funneling","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02069","snapshot_observed_at":"2026-08-04T07:44:17.063957Z","title":"Pyramidkv: Dynamic kv cache compression based on pyramidal information funneling.arXiv preprint arXiv:2406.02069,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:17.063957Z"},"links":{"cited_paper":"/paper/2406.02069","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:75a8c9bfebe4315f156f7f0753cad023fbc298818785ad9c1c9a1bc9a7d03f6a","observation_id":"095e062d-1ca3-47bf-a3b7-d314d4b98893","resolution":{"observed_at":"2026-08-04T07:44:17.063957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T07:44:17.215338Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:17.215338Z"},"links":{"citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:91e4354f8d4e766438e47f725841641ac2cf82984dd17bf54d65309e110f7e2f","observation_id":"136bda79-b2af-44c4-a312-226f3b42fda4","resolution":{"observed_at":"2026-08-04T07:44:17.215338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T07:44:17.535583Z","title":"On Gemma2-2b-it, retaining the top 1k tokens per layer, DHSA matches dense attention while substantially outperforming block sparse attention","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:17.535583Z"},"links":{"citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:f346d8d50443882e6e19d69756e308cafc7285d8a4153478fc819396493b9126","observation_id":"1b78366d-7034-4a81-b543-ad98329af1f6","resolution":{"observed_at":"2026-08-04T07:44:17.535583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T07:44:17.684744Z","title":"Sliding-window attention uses a budget of 2048, block sparse attention 512 and KV compression (streaming LLM, h2o, pyramidKV)","venue":null,"work_id":null,"year":2048},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:17.684744Z"},"links":{"citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:74c3ceef6ad033676b792bcce87beda461a762311ae6df64e4eb01e456e7f918","observation_id":"1d36bdba-36ce-4b52-99c2-b9026f914c84","resolution":{"observed_at":"2026-08-04T07:44:17.684744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T07:44:17.964749Z","title":"Sliding- window attention uses a budget of 2048, block sparse attention 512 and KV compression (streaming LLM, h2o, pyramidKV)","venue":null,"work_id":null,"year":2048},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:17.964749Z"},"links":{"citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:181174a33f36d05b231d7930c25926bbe3118b6d91c17ae82e7a1b4c6a525983","observation_id":"54711909-d2fc-45da-af4c-e45b342aa9e8","resolution":{"observed_at":"2026-08-04T07:44:17.964749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T07:44:18.183182Z","title":"Finally, we note several potential influencing factors in KV compression methods (Fig","venue":null,"work_id":null,"year":2048},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:18.183182Z"},"links":{"citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:ceb4942050c4aa060f4508b4e0749e161b575b8eb1a68e6da8878043432adb30","observation_id":"3fe1254e-71a7-41bc-a966-74fbff2ef4e1","resolution":{"observed_at":"2026-08-04T07:44:18.183182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T07:44:18.354744Z","title":"0 2000 4000 6000 8000 10000 12000 14000 16000 Max KV Cache Capacity 3.2 3.4 3.6 3.8 4.0 4.2Time (s) Latency v.s","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:18.354744Z"},"links":{"citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:7f5b483d601665cb305df86a79ae5c5dd3b87df9251175bdb9cde424f15d0eb1","observation_id":"87125184-9e45-400c-8394-5f47199d66a5","resolution":{"observed_at":"2026-08-04T07:44:18.354744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T07:44:17.604884Z","title":"On the other hand, for all decode-stage methods, prefill time increases with context length since KV cache compression only applies during decoding","venue":null,"work_id":null,"year":2048},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:17.604884Z"},"links":{"citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:0baeebf805b45bb2de8654b9893749adb79deea2108bbc9e526908f2dcb2ab5b","observation_id":"a0d3ba07-b0ee-4528-9f43-3cea5950c520","resolution":{"observed_at":"2026-08-04T07:44:17.604884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17453","last_updated":"2024-04-07T00:56:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-29T17:59:56Z","title":"Efficient Streaming Language Models with Attention Sinks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17453","snapshot_observed_at":"2026-08-04T07:44:16.904840Z","title":"Efficient streaming language models with attention sinks.arXiv preprint arXiv:2309.17453,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":2006,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:16.904840Z"},"links":{"cited_paper":"/paper/2309.17453","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:93f4fcc79d8c2f396f2e6e81f1ab2ec39ffebe61041147eb89f2dbd805afd434","observation_id":"9c7d66e6-b2cb-45c8-a476-0885ed71c286","resolution":{"observed_at":"2026-08-04T07:44:16.904840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-08-04T07:44:15.991143Z","title":"Longformer: The long-document transformer","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:15.991143Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:4f27c7c2c36a7c7d89494e62334592870b52381868567c18ad4165b3df7d049b","observation_id":"1a44c33b-1804-44a1-9a42-835bc06d0035","resolution":{"observed_at":"2026-08-04T07:44:15.991143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14508","last_updated":"2024-06-19T04:00:32Z","snapshot_observed_at":"2026-08-02T11:20:36.216220Z","submitted_at":"2023-08-28T11:53:40Z","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14508","snapshot_observed_at":"2026-08-04T07:44:16.274893Z","title":"Longbench: A bilingual, multitask benchmark for long context understanding.arXiv preprint arXiv:2308.14508,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:16.274893Z"},"links":{"cited_paper":"/paper/2308.14508","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:429732d24eb300edcb19acec0a1febff24b8f64f18594d9c3e2e6c908fe2f06d","observation_id":"9578a1aa-a2c1-4bc3-97ce-afc38ff1ad6c","resolution":{"observed_at":"2026-08-04T07:44:16.274893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16137","last_updated":"2024-06-24T21:22:00Z","snapshot_observed_at":"2026-08-07T02:35:31.570950Z","submitted_at":"2023-08-30T16:47:51Z","title":"LM-Infinite: Zero-Shot Extreme Length Generalization for Large Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16137","snapshot_observed_at":"2026-08-04T07:44:16.124951Z","title":"Lm- infinite: Zero-shot extreme length generalization for large language models.arXiv preprint arXiv:2308.16137,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:16.124951Z"},"links":{"cited_paper":"/paper/2308.16137","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:0fd4c57f421e6ec8912ed007fad79c48b6c99e9b14265f6cdb5dc467990e2e4e","observation_id":"d1f67058-5aeb-4794-9502-cf08d945e7e4","resolution":{"observed_at":"2026-08-04T07:44:16.124951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1705.03551","last_updated":"2017-05-13T21:12:37Z","snapshot_observed_at":"2026-08-02T11:13:42.401488Z","submitted_at":"2017-05-09T21:35:07Z","title":"TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.03551","snapshot_observed_at":"2026-08-04T07:44:16.653934Z","title":"triviaqa: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension.arXiv e-prints, art","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-04T07:44:16.653934Z"},"links":{"cited_paper":"/paper/1705.03551","citing_paper":"/paper/2510.24606"},"observation_digest":"sha256:0b8dd0a0696cc4c809b4b4ab07c276e6ecb40fd6e698b6415b71156c9574d9f1","observation_id":"1c2aa70a-518a-4ec3-b830-f9a654911648","resolution":{"observed_at":"2026-08-04T07:44:16.653934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2510.24606","last_updated":"2026-05-27T23:25:55Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-04T07:44:14.501838Z","submitted_at":"2025-10-28T16:34:18Z","title":"Long-Context Modeling with Dynamic Hierarchical Sparse Attention for Memory-Constrained LLM Inference"},"reference_resolution":{"displayed":17,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":17},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 17 of 17 outbound references and 0 inbound Pith citation observations for arXiv:2510.24606."}