{"as_of":"2026-08-09T21:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1d8390292532b350d1ddb1f1a6bffa98a8c19496bb2f9eff962f7c1b424e1e22","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":9,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":9,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T00:41:45.459582Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T23:27:28.048568Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2504.09844","last_updated":"2026-04-27T14:30:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T03:31:22Z","title":"MegaScale-Data: Scaling Dataloader for Multisource Large Foundation Model Training","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-22T21:12:22.201810Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2504.09844"},"observation_digest":"sha256:7e4d54a2a2009b28f9cea646467dfd47e01a1c7fbd4c8b47658d61a67bd5c51d","observation_id":"8017b579-2402-4f6b-95f9-c90fcc2523bd","resolution":{"observed_at":"2026-05-22T21:15:09.582899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2505.13211","last_updated":"2025-05-19T14:58:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-19T14:58:50Z","title":"MAGI-1: Autoregressive Video Generation at Scale","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T20:31:15.700943Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2505.13211"},"observation_digest":"sha256:c1ca4e78c92633232bea3e46154cf3a60dd2ec0459c4aaba43be0ce1c3809629","observation_id":"d062ed69-fb33-4f4d-9da4-7b47739030ca","resolution":{"observed_at":"2026-05-13T20:31:15.775808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2509.21275","last_updated":"2026-04-25T07:48:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-25T15:01:25Z","title":"InfiniPipe: Elastic Pipeline Parallelism for Efficient Variable-Length Long-Context LLM Training","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T14:04:31.017142Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2509.21275"},"observation_digest":"sha256:bff488296d0791073b71bf16632822a0c961187dac275efa1e3685a16d056001","observation_id":"033a95d0-8d15-4b90-991a-67655237a820","resolution":{"observed_at":"2026-05-18T14:06:27.261784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2510.18830","last_updated":"2026-05-19T17:27:20Z","snapshot_observed_at":"2026-07-06T22:33:43.999976Z","submitted_at":"2025-10-21T17:25:32Z","title":"MTraining: Distributed Dynamic Sparse Attention for Efficient Ultra-Long Context Training","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-21T19:44:04.833504Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2510.18830"},"observation_digest":"sha256:44150eeff3af4198a270d198a52cf88b49044a4421d0a56b8ea5e8bda9622864","observation_id":"a492123d-5b69-401e-b410-6bbfaff96408","resolution":{"observed_at":"2026-05-21T19:44:19.608821Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2602.15763","last_updated":"2026-02-24T10:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-17T17:50:56Z","title":"GLM-5: from Vibe Coding to Agentic Engineering","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T05:46:40.836161Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2602.15763"},"observation_digest":"sha256:c43ffb56a483111d324af42504cf3ba5c6f67f927cd339f81674922159de1b8c","observation_id":"dc9975fe-da9b-44d6-a3bd-52eb37215dba","resolution":{"observed_at":"2026-05-11T05:46:40.953528Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2604.21026","last_updated":"2026-04-24T07:54:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-22T19:18:30Z","title":"MCAP: Deployment-Time Layer Profiling for Memory-Constrained LLM Inference","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T00:54:33.897112Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2604.21026"},"observation_digest":"sha256:a23a560ee368b1903296aa7fbbbf6736077a1edb91c52b52013927ff6d3b9d45","observation_id":"e5ff70bb-1d1b-4c95-87ef-838c8574ecd0","resolution":{"observed_at":"2026-05-10T00:54:48.395862Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2605.08962","last_updated":"2026-05-09T13:59:27Z","snapshot_observed_at":"2026-07-06T23:21:06.773761Z","submitted_at":"2026-05-09T13:59:27Z","title":"MegaScale-Omni: A Hyper-Scale, Workload-Resilient System for MultiModal LLM Training in Production","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T02:04:07.344134Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2605.08962"},"observation_digest":"sha256:ac83e498107e7c796b81251863f32b10e7d729eeb4581471accd6811cc40a6a1","observation_id":"c1b504b3-a5c5-42c3-8af3-6e14e81c1f74","resolution":{"observed_at":"2026-05-12T02:06:14.982178Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":"2502.21231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-07-02T23:27:28.048568Z","title":"Bytescale: Efficient scaling of llm training with a 2048k context length on more than 12,000 gpus","venue":null,"work_id":"660685cd-8bce-447d-a0db-b0a85fc5b20d","year":2025},"citing_paper":{"arxiv_id":"2606.08476","last_updated":"2026-06-07T06:45:15Z","snapshot_observed_at":"2026-08-01T22:16:48.770899Z","submitted_at":"2026-06-07T06:45:15Z","title":"FlashCP: Load-Balanced Communication-Efficient Context Parallelism for LLM Training","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T18:10:34.962717Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2606.08476"},"observation_digest":"sha256:ad6d2f1268902a8e3221f2ebf0a4ef330052a5f90d9b5e677005f331370a1170","observation_id":"e25cb165-2ed7-4dd4-b56c-c6e1d6cf8167","resolution":{"observed_at":"2026-07-02T23:27:28.050061Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.21231","snapshot_observed_at":"2026-08-02T00:41:45.459582Z","title":"GLM-5-Team","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14952","last_updated":"2026-07-27T13:07:41Z","snapshot_observed_at":"2026-08-08T19:48:49.069016Z","submitted_at":"2026-07-16T13:00:32Z","title":"LongStraw: Long-Context RL Beyond 2M Tokens under a Fixed GPU Budget","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T00:41:45.459582Z"},"links":{"cited_paper":"/paper/2502.21231","citing_paper":"/paper/2607.14952"},"observation_digest":"sha256:0f5424adc4359b0bece97d55efb77a9ab926b821b07aa4da81b7d0a91fbacce3","observation_id":"4e602654-14e9-497d-ab0b-7d5d49199b30","resolution":{"observed_at":"2026-08-02T00:41:45.459582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.21231/citation-record","integrity":"/paper/2502.21231/integrity","json":"/paper/2502.21231/citation-record.json","paper":"/paper/2502.21231"},"outbound":[],"paper":{"arxiv_id":"2502.21231","last_updated":"2025-02-28T17:01:03Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-07T17:39:08.151477Z","submitted_at":"2025-02-28T17:01:03Z","title":"ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 9 inbound Pith citation observations for arXiv:2502.21231."}