{"as_of":"2026-08-10T08:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4f084fb8215b6a0ea4506b0fa0a7c30b2d582123ba676bc99b485a46301d32ce","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":41,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":41,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":41,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T11:50:03.340792Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-13T05:47:27.775529Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2403.04652"},"observation_digest":"sha256:9d2c28f89f8c3d98f3d4649e2955bfed55f7cae520a056ed1483434fc629329f","observation_id":"de159343-96e4-44b2-b362-1894086c305c","resolution":{"observed_at":"2026-05-13T05:47:27.926738Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2403.17297","last_updated":"2024-03-26T00:53:24Z","snapshot_observed_at":"2026-08-02T11:10:24.263044Z","submitted_at":"2024-03-26T00:53:24Z","title":"InternLM2 Technical Report","version":1},"reference_index":151,"source":"arxiv_source","source_observed_at":"2026-05-15T11:44:38.066501Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2403.17297"},"observation_digest":"sha256:dca57e69e57bdcf39795f2782dfcf2fb2b4b22f6640e498c254ad7b0e8cf5d38","observation_id":"b751f984-93be-4913-a9ad-b0e1f11f1341","resolution":{"observed_at":"2026-05-15T11:44:38.193152Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2403.20208","last_updated":"2026-04-22T07:34:25Z","snapshot_observed_at":"2026-07-06T17:53:06.519891Z","submitted_at":"2024-03-29T14:41:21Z","title":"Unlock the Potential of Large Language Models for Predictive Tabular Tasks in Data Science with Table-Specific Pretraining","version":8},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-24T02:44:01.340415Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2403.20208"},"observation_digest":"sha256:4a630b085d5d2afde0a0366c7c282fe16b8c77080ece9d7d8b96e17b06248b06","observation_id":"070d67f6-37b9-4881-89a9-033a825c40db","resolution":{"observed_at":"2026-05-24T02:45:56.148382Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2404.06395","last_updated":"2024-06-03T08:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-09T15:36:50Z","title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T18:00:53.389420Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2404.06395"},"observation_digest":"sha256:3893070dc969775163242aae04cbd18c8c2b0873915d0594a94b8d5a6419452b","observation_id":"b7e885b4-9ee4-48ac-9fc8-46f35fdd1af0","resolution":{"observed_at":"2026-05-13T18:00:53.587242Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2404.07143","last_updated":"2024-08-09T22:37:25Z","snapshot_observed_at":"2026-08-02T14:27:24.212588Z","submitted_at":"2024-04-10T16:18:42Z","title":"Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-21T18:17:00.157129Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2404.07143"},"observation_digest":"sha256:5f388a9052ae0525d87c80ef569d47292ba93530181a098d4da1457752397ebe","observation_id":"2e04b2ef-fc49-44a2-80dd-3f3838c390c4","resolution":{"observed_at":"2026-05-21T18:17:00.182745Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2406.07887","last_updated":"2024-06-12T05:25:15Z","snapshot_observed_at":"2026-07-06T18:29:21.709395Z","submitted_at":"2024-06-12T05:25:15Z","title":"An Empirical Study of Mamba-based Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-18T10:31:03.777169Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2406.07887"},"observation_digest":"sha256:dea10d710e1685910bd8e03b43a58f7c4368352a95527dac5fedfbd60cce7895","observation_id":"b8fb69a8-2b39-455d-b764-0f6d566955ae","resolution":{"observed_at":"2026-05-18T10:31:03.937188Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"reference_index":205,"source":"pdf_text","source_observed_at":"2026-05-17T22:58:16.523267Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2406.11794"},"observation_digest":"sha256:adf97cdc60a2bd25ad589a6dafb3ea441cdb46e69f4db96d0fbb26e09be173c8","observation_id":"a1f18415-5f96-4a5d-b965-1694933d7121","resolution":{"observed_at":"2026-05-17T22:58:17.328146Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-11T08:08:09.444352Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2406.12793"},"observation_digest":"sha256:70eb2cd2b8ffdbda8832a75faa0ad868376fbe08addecf8c1e0c5b662d2a3501","observation_id":"1efb07d9-cfa6-45f8-8402-3048e9315387","resolution":{"observed_at":"2026-05-11T08:08:09.592936Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2408.06072","last_updated":"2025-03-26T08:33:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-12T11:47:11Z","title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-10T18:26:22.224924Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2408.06072"},"observation_digest":"sha256:de57d4bcbb675affa4a44a0c06758e39cff088e9a8ce73099fa7d3580f4f6873","observation_id":"ceaa4416-f71c-4ce5-b43a-767c2f1d8ec2","resolution":{"observed_at":"2026-05-10T18:26:22.274236Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-23T06:25:00.376073Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2412.15115"},"observation_digest":"sha256:4681557464254db56aa485d8990ecb91b2c67b27c79ba1c24c4556e48949c32c","observation_id":"24f88f60-4a42-4d81-8380-fccf16f39c25","resolution":{"observed_at":"2026-05-23T06:25:27.807139Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2501.15383","last_updated":"2025-01-26T03:47:25Z","snapshot_observed_at":"2026-07-31T01:49:29.562668Z","submitted_at":"2025-01-26T03:47:25Z","title":"Qwen2.5-1M Technical Report","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-15T05:25:59.964619Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2501.15383"},"observation_digest":"sha256:aa730f5df3a9530831f12bf52023bc8492676195336d4372458f935e7b425c3a","observation_id":"95c1ff53-a00f-4553-93c3-e7972323f675","resolution":{"observed_at":"2026-05-15T05:26:00.157889Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-09T11:50:03.340792Z","title":"Effective long-context scaling of foundation models.arXiv preprint arXiv:2309.16039, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.02562","last_updated":"2025-02-04T18:37:17Z","snapshot_observed_at":"2026-08-09T11:42:18.491712Z","submitted_at":"2025-02-04T18:37:17Z","title":"Learning the RoPEs: Better 2D and 3D Position Encodings with STRING","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-09T11:50:03.340792Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2502.02562"},"observation_digest":"sha256:6fda4d1d35b1a82ff7e06b16653f831aad2b29444f42fdc04368596f151e0fdf","observation_id":"deef4fd9-70e1-4e17-9a3e-b4ff77da508b","resolution":{"observed_at":"2026-08-09T11:50:03.340792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-09T11:11:17.807536Z","title":"A., Oguz, B., Khabsa, M., Fang, H., Mehdad, Y., Narang, S., Malik, K., Fan, A., Bhosale, S., Edunov, S., Lewis, M., Wang, S., and Ma, H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.02789","last_updated":"2025-05-19T18:24:50Z","snapshot_observed_at":"2026-08-10T06:31:01.602594Z","submitted_at":"2025-02-05T00:22:06Z","title":"Speculative Prefill: Turbocharging TTFT with Lightweight and Training-Free Token Importance Estimation","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-09T11:11:17.807536Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2502.02789"},"observation_digest":"sha256:c6b34367d1b7b7e7a45db9a797cf51dd88e40229d8b24a652d49702e9afcbc35","observation_id":"f80f3215-5206-4997-ac88-d9ea0cd44b21","resolution":{"observed_at":"2026-08-09T11:11:17.807536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-09T05:48:07.677274Z","title":"A., Oguz, B., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.03147","last_updated":"2025-02-05T13:16:41Z","snapshot_observed_at":"2026-08-10T03:55:45.471524Z","submitted_at":"2025-02-05T13:16:41Z","title":"Scalable In-Context Learning on Tabular Data via Retrieval-Augmented Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-09T05:48:07.677274Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2502.03147"},"observation_digest":"sha256:c91f71c797f7998612a966b8f1197295e366935868473dab8c6f50ceb8f7019c","observation_id":"c75a5024-6706-41a3-a1ff-a6369bb8ca60","resolution":{"observed_at":"2026-08-09T05:48:07.677274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-09T05:02:35.824173Z","title":"A., Oguz, B., Khabsa, M., Fang, H., Mehdad, Y., Narang, S., Malik, K., Fan, A., Bhosale, S., Edunov, S., Lewis, M., Wang, S., and Ma, H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.03358","last_updated":"2025-06-09T14:31:04Z","snapshot_observed_at":"2026-08-09T13:31:46.100534Z","submitted_at":"2025-02-05T16:53:45Z","title":"Minerva: A Programmable Memory Test Benchmark for Language Models","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-09T05:02:35.824173Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2502.03358"},"observation_digest":"sha256:4a4de6c45fd83fc1f417426cd8a786dc7ba5ce8b8a556515800ed98cb76c6c06","observation_id":"39c83ac0-c8f2-4c4c-b025-86d7bc3dfe3d","resolution":{"observed_at":"2026-08-09T05:02:35.824173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-08T11:20:23.698827Z","title":"org/abs/2309.16039","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.07972","last_updated":"2025-03-09T19:39:00Z","snapshot_observed_at":"2026-08-09T03:34:52.654329Z","submitted_at":"2025-02-11T21:36:31Z","title":"Training Sparse Mixture Of Experts Text Embedding Models","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T11:20:23.698827Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2502.07972"},"observation_digest":"sha256:1b756ecbcfcf75312f981a057125f4c6d26bd9808a4b717bb16f5255d867b4b2","observation_id":"87cf181b-9fce-4008-b8a2-ed9fabef6bc7","resolution":{"observed_at":"2026-08-08T11:20:23.698827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2502.13189","last_updated":"2025-02-18T14:06:05Z","snapshot_observed_at":"2026-07-06T20:38:48.725605Z","submitted_at":"2025-02-18T14:06:05Z","title":"MoBA: Mixture of Block Attention for Long-Context LLMs","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-16T06:15:46.085555Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2502.13189"},"observation_digest":"sha256:da97b51741cd0bd8498bada67ca4e0ebe18240b9c310e7ef1fc9405b06e57628","observation_id":"3de63877-21a0-4fe7-b85d-29e2c110e9a0","resolution":{"observed_at":"2026-05-16T06:15:46.122456Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-09T06:35:27.813995Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2505.09388"},"observation_digest":"sha256:41361076aa0a7d90fc977aae7060ac5cd1d12f071f0961dbdab430fd5ed278b2","observation_id":"a8d8f529-d730-47bb-b74f-28e120acbda6","resolution":{"observed_at":"2026-05-09T06:35:28.628333Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-07T15:08:03.619535Z","title":"Effective long-context scaling of foundation models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17134","last_updated":"2025-06-03T03:04:17Z","snapshot_observed_at":"2026-08-09T07:25:09.275709Z","submitted_at":"2025-05-22T04:05:02Z","title":"LongMagpie: A Self-synthesis Method for Generating Large-scale Long-context Instructions","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:03.619535Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2505.17134"},"observation_digest":"sha256:2b2787e381b9584b312aef05430856ace4ffc7c46ce801dd2508253365165b7e","observation_id":"ac372b97-7aa7-4b73-a546-883bb3bd8fc3","resolution":{"observed_at":"2026-08-07T15:08:03.619535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-07T14:52:14.996527Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17296","last_updated":"2025-05-22T21:23:20Z","snapshot_observed_at":"2026-08-09T05:52:01.092080Z","submitted_at":"2025-05-22T21:23:20Z","title":"SELF: Self-Extend the Context Length With Logistic Growth Function","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:14.996527Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2505.17296"},"observation_digest":"sha256:4aeb5b4e673726e39f2c3e022f0399cec0f517e93579368c6351ceadc6e5691b","observation_id":"498644d0-ddce-4eef-a8db-a0c47deab8f7","resolution":{"observed_at":"2026-08-07T14:52:14.996527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-07T14:52:55.492170Z","title":"Effective long-context scaling of foundation models.arXiv preprint arXiv:2309.16039, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17315","last_updated":"2026-06-02T22:13:24Z","snapshot_observed_at":"2026-08-07T22:00:02.564609Z","submitted_at":"2025-05-22T22:09:47Z","title":"Longer Context, Deeper Thinking: Uncovering the Role of Long-Context Ability in Reasoning","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:52:55.492170Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2505.17315"},"observation_digest":"sha256:fdd4f55d8df7ef8792fc19ee1ebe4b27251472c82a86dd8df82bf591e5f771ad","observation_id":"1b9824f0-230c-4c10-8cb7-1ebe1d5bdeae","resolution":{"observed_at":"2026-08-07T14:52:55.492170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-07T14:20:26.452682Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19293","last_updated":"2026-06-02T22:08:40Z","snapshot_observed_at":"2026-08-07T14:15:05.692688Z","submitted_at":"2025-05-25T19:58:31Z","title":"100-LongBench: Are de facto Long-Context Benchmarks Literally Evaluating Long-Context Ability?","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T14:20:26.452682Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2505.19293"},"observation_digest":"sha256:43978e3ea822d1bc654d031eed5d746d70b42aa0190179cbd9a1aaa3ed6654b1","observation_id":"b4197dba-3d4d-4a2f-9023-2461b166a976","resolution":{"observed_at":"2026-08-07T14:20:26.452682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-07T04:54:26.223746Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09440","last_updated":"2025-06-11T06:46:49Z","snapshot_observed_at":"2026-08-09T03:35:26.746241Z","submitted_at":"2025-06-11T06:46:49Z","title":"GigaChat Family: Efficient Russian Language Modeling Through Mixture of Experts Architecture","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T04:54:26.223746Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2506.09440"},"observation_digest":"sha256:046213434ce9f9ef01513064dd37463a31a92fa0c3b0489b14d59055dcc13a3a","observation_id":"5ea00957-85e6-4111-9afa-d33b3265181d","resolution":{"observed_at":"2026-08-07T04:54:26.223746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-06T23:49:48.167491Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16024","last_updated":"2025-06-19T04:44:34Z","snapshot_observed_at":"2026-08-09T12:39:00.788337Z","submitted_at":"2025-06-19T04:44:34Z","title":"From General to Targeted Rewards: Surpassing GPT-4 in Open-Ended Long-Context Generation","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T23:49:48.167491Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2506.16024"},"observation_digest":"sha256:bfe26eb5a08ce1bbf71648c731d40184c8f47efaa48bd560e82989ff9ba20fbe","observation_id":"1af95218-cebe-4276-b85b-922499cfaf73","resolution":{"observed_at":"2026-08-06T23:49:48.167491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2507.02259","last_updated":"2026-07-29T12:55:39Z","snapshot_observed_at":"2026-08-06T20:31:23.587108Z","submitted_at":"2025-07-03T03:11:50Z","title":"MemAgent: Reshaping Long-Context LLM with Multi-Conv RL-based Memory Agent","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T11:17:24.406028Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2507.02259"},"observation_digest":"sha256:75176b37515725f34615210041fbb5690aedfb99657bc78313d18a838064b987","observation_id":"f15daf29-d480-49e0-9f74-7e0249d6a1c1","resolution":{"observed_at":"2026-05-15T11:17:24.561802Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-06T20:40:09.893452Z","title":"Effective long-context scaling of foundation models.arXiv preprint arXiv:2309.16039, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02259","last_updated":"2026-07-29T12:55:39Z","snapshot_observed_at":"2026-08-06T20:31:23.587108Z","submitted_at":"2025-07-03T03:11:50Z","title":"MemAgent: Reshaping Long-Context LLM with Multi-Conv RL-based Memory Agent","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:09.893452Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2507.02259"},"observation_digest":"sha256:dffd16e4e6d2db716ccbfa112ca18c5105e2711e3e018c14177fe90550da67a2","observation_id":"f97edf73-0655-4743-9206-f8c70b20bf38","resolution":{"observed_at":"2026-08-06T20:40:09.893452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-06T05:13:27.252272Z","title":"Effective long-context scaling of foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.02122","last_updated":"2025-08-04T07:03:35Z","snapshot_observed_at":"2026-08-09T16:03:00.284372Z","submitted_at":"2025-08-04T07:03:35Z","title":"An Overview of Algorithms for Contactless Cardiac Feature Extraction from Radar Signals: Advances and Challenges","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T05:13:27.252272Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2508.02122"},"observation_digest":"sha256:46b07ac7dbe95c6fb2e7a268913973ebe5ee7238b583a25c199853f44e38f67a","observation_id":"0578d433-6bfb-4e80-a395-496323676242","resolution":{"observed_at":"2026-08-06T05:13:27.252272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T17:46:46.815258Z","title":"Effective long-context scaling of foundation models.arXiv preprint arXiv:2309.16039,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.15717","last_updated":"2025-08-21T16:56:29Z","snapshot_observed_at":"2026-08-09T15:11:23.510316Z","submitted_at":"2025-08-21T16:56:29Z","title":"StreamMem: Query-Agnostic KV Cache Memory for Streaming Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T17:46:46.815258Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2508.15717"},"observation_digest":"sha256:34a3df9ddb17bad44230d28518d1401060643b2318c545bb58a05d8cb623a2db","observation_id":"400ee6ed-76b1-4e79-a080-bb9d991620fb","resolution":{"observed_at":"2026-08-05T17:46:46.815258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T17:26:28.943342Z","title":"Effective long-context scaling of foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.16279","last_updated":"2025-08-22T10:35:56Z","snapshot_observed_at":"2026-08-07T15:35:45.566152Z","submitted_at":"2025-08-22T10:35:56Z","title":"AgentScope 1.0: A Developer-Centric Framework for Building Agentic Applications","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-05T17:26:28.943342Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2508.16279"},"observation_digest":"sha256:183a2df376be19abe90abca6ef5928c3de1dfbe7bd1b2036ee255758fecfae5a","observation_id":"bf315b5a-5084-4d13-88b4-86c9807bab4f","resolution":{"observed_at":"2026-08-05T17:26:28.943342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T05:36:56.768551Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.05218","last_updated":"2025-09-08T03:13:38Z","snapshot_observed_at":"2026-08-09T17:41:45.675571Z","submitted_at":"2025-09-05T16:20:48Z","title":"HoPE: Hyperbolic Rotary Positional Encoding for Stable Long-Range Dependency Modeling in Large Language Models","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T05:36:56.768551Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2509.05218"},"observation_digest":"sha256:34ca8ce7412bc853d6975fad931aedaaa489f615b51c63c40b66e897e124e1ef","observation_id":"108ef45b-43d2-4de8-86ff-1f5f4fc182d4","resolution":{"observed_at":"2026-08-05T05:36:56.768551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-04T21:37:23.079381Z","title":"Effective long-context scaling of foundation models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08032","last_updated":"2025-09-09T16:09:19Z","snapshot_observed_at":"2026-08-08T14:11:18.383165Z","submitted_at":"2025-09-09T16:09:19Z","title":"SciGPT: A Large Language Model for Scientific Literature Understanding and Knowledge Discovery","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T21:37:23.079381Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2509.08032"},"observation_digest":"sha256:b0e34036900f98f2bda1bd29e20e0599eeb5d33f43a513e4df56ff45503e29f6","observation_id":"fd421750-e2d4-4161-86ff-1640931f675c","resolution":{"observed_at":"2026-08-04T21:37:23.079381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2509.12635","last_updated":"2026-05-10T22:28:15Z","snapshot_observed_at":"2026-08-03T03:02:00.357806Z","submitted_at":"2025-09-16T03:53:32Z","title":"Positional Encoding via Token-Aware Phase Attention","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-18T16:19:48.318702Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2509.12635"},"observation_digest":"sha256:07e2b4e06902eda064d1765d0e2f39a8d754a28a3181f3ff53637cb8d51cc6a3","observation_id":"68fe0276-efc3-4a0d-8d17-01c722473bd1","resolution":{"observed_at":"2026-05-18T16:21:36.709109Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2510.26692","last_updated":"2025-11-01T12:05:18Z","snapshot_observed_at":"2026-08-07T19:30:34.681869Z","submitted_at":"2025-10-30T16:59:43Z","title":"Kimi Linear: An Expressive, Efficient Attention Architecture","version":2},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-05-13T23:49:10.555255Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2510.26692"},"observation_digest":"sha256:83ab2acd96ec348dc3885c86695d117d77700ab7ba84034edcd6d31bd35779f1","observation_id":"54d2f944-117c-4f58-9909-61e7e9fba39c","resolution":{"observed_at":"2026-05-13T23:49:10.920344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2512.07805","last_updated":"2026-05-13T18:33:36Z","snapshot_observed_at":"2026-07-06T22:38:08.821098Z","submitted_at":"2025-12-08T18:39:13Z","title":"Group Representational Position Encoding","version":6},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T00:04:13.707931Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2512.07805"},"observation_digest":"sha256:fcd32d864590489c2fb890ba8c5a55dea59f2454a2e2fbcecffaaa791cab6ed3","observation_id":"4da40a99-a044-43fd-8821-49234d6547b3","resolution":{"observed_at":"2026-05-17T00:08:43.795033Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2603.23885","last_updated":"2026-04-18T12:13:54Z","snapshot_observed_at":"2026-07-06T22:50:28.640006Z","submitted_at":"2026-03-25T03:19:09Z","title":"Towards Real-World Document Parsing via Realistic Scene Synthesis and Document-Aware Training","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-15T01:15:26.757215Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2603.23885"},"observation_digest":"sha256:132ac7e120df9fb6a189395cf1b81da637d2e3e75b716f344e36a9d5feb9d611","observation_id":"887d0e6b-6dd6-4ffb-81b2-2f30da466347","resolution":{"observed_at":"2026-05-15T01:18:26.583970Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2604.07809","last_updated":"2026-04-09T05:07:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-09T05:07:57Z","title":"PolicyLong: Towards On-Policy Context Extension","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T17:23:28.939977Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2604.07809"},"observation_digest":"sha256:d8c1d3004166176d02c6f6dcdb4be32ab905aecf253a891e0c14c89d65f4a929","observation_id":"360a7722-9923-41b7-aa4a-43901ae9cbcf","resolution":{"observed_at":"2026-05-11T06:55:59.842670Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2604.16864","last_updated":"2026-04-18T06:28:21Z","snapshot_observed_at":"2026-08-02T05:52:50.560554Z","submitted_at":"2026-04-18T06:28:21Z","title":"HieraSparse: Hierarchical Semi-Structured Sparse KV Attention","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T07:15:19.184970Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2604.16864"},"observation_digest":"sha256:d91ca379cbd8630a8a52b51540513fbb16847b0c8c139d0863db2a0a8e31f145","observation_id":"3dec61b5-ca15-4c0b-b0e3-e15ef88e1595","resolution":{"observed_at":"2026-05-10T07:16:54.154856Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2606.05345","last_updated":"2026-06-19T19:44:08Z","snapshot_observed_at":"2026-07-06T23:45:23.749718Z","submitted_at":"2026-06-03T18:37:46Z","title":"PJ-RoPE: A Fourier-Jet-Affine Position Space for Relative Attention","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T07:22:20.966483Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2606.05345"},"observation_digest":"sha256:65b70caf28e262805557b7729381693e51f98bb55af1dab73359657e47eaa650","observation_id":"2af03048-6b25-4f6e-b3bc-93e7fba983f3","resolution":{"observed_at":"2026-07-02T06:36:44.000716Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2606.30460","last_updated":"2026-06-30T09:40:56Z","snapshot_observed_at":"2026-08-01T15:05:23.676611Z","submitted_at":"2026-06-29T15:26:55Z","title":"HSAP: A Hierarchical Sequence-aware Parallelism for Hybrid-Context Generative Models","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-06-30T07:27:34.161517Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2606.30460"},"observation_digest":"sha256:1d29e607be51cec3dc9b15d56098f061adae855cd6700b885c4ccd112a7b74db","observation_id":"9a93caa2-d800-4c6c-9803-82d516048794","resolution":{"observed_at":"2026-06-30T07:34:21.743679Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":"2309.16039","doi":"10.48550/arxiv.2309.16039","metadata_source":"arxiv_reference","pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Oguz, B., et al","venue":"arXiv (Cornell University)","work_id":"100ca4eb-be6f-4510-9c7f-bbb24f10d946","year":2023},"citing_paper":{"arxiv_id":"2606.30460","last_updated":"2026-06-30T09:40:56Z","snapshot_observed_at":"2026-08-01T15:05:23.676611Z","submitted_at":"2026-06-29T15:26:55Z","title":"HSAP: A Hierarchical Sequence-aware Parallelism for Hybrid-Context Generative Models","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-07-01T06:58:30.817223Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2606.30460"},"observation_digest":"sha256:6fe43e8f0a681b6f371da04a045ad058077a523440413ffc2a2302cfe7d03d50","observation_id":"5cd62d5c-7320-4fd9-b90b-e19165b97681","resolution":{"observed_at":"2026-07-01T08:55:35.666180Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16039","snapshot_observed_at":"2026-08-02T14:07:53.521077Z","title":"Effective","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.20448","last_updated":"2026-05-13T15:26:50Z","snapshot_observed_at":"2026-08-10T04:23:26.799831Z","submitted_at":"2026-05-13T15:26:50Z","title":"Domyn-Small: A European 10B Reasoning Language Model","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-02T14:07:53.521077Z"},"links":{"cited_paper":"/paper/2309.16039","citing_paper":"/paper/2607.20448"},"observation_digest":"sha256:14b1c150448778e3dfac678123cf7cf64d0665d6d0790de8ebfdeef80f2b1a89","observation_id":"938a1682-9e9e-4d56-8ed0-cfc9c55d629a","resolution":{"observed_at":"2026-08-02T14:07:53.521077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2309.16039/citation-record","integrity":"/paper/2309.16039/integrity","json":"/paper/2309.16039/citation-record.json","paper":"/paper/2309.16039"},"outbound":[],"paper":{"arxiv_id":"2309.16039","last_updated":"2023-11-14T01:40:13Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:24:38.375055Z","submitted_at":"2023-09-27T21:41:49Z","title":"Effective Long-Context Scaling of Foundation Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 41 inbound Pith citation observations for arXiv:2309.16039."}