{"as_of":"2026-08-09T05:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3493262e554a0e13e07fc6085a6f3ca6716c7cda8624bcf858471c64d3730083","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":26,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T13:44:00.474327Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":24,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2306.01116","last_updated":"2023-06-01T20:03:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-01T20:03:56Z","title":"The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data, and Web Data Only","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T20:43:45.770157Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2306.01116"},"observation_digest":"sha256:6b41a11fcc1fbb3fc06b1d4439ab755df02bfacd906f549c3461a602ec6a9b05","observation_id":"72709979-5cf3-409f-93c8-87bf672b1828","resolution":{"observed_at":"2026-05-13T20:43:45.805619Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2311.16867","last_updated":"2023-11-29T19:45:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-28T15:12:47Z","title":"The Falcon Series of Open Language Models","version":2},"reference_index":273,"source":"arxiv_source","source_observed_at":"2026-05-16T09:46:09.701440Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2311.16867"},"observation_digest":"sha256:278401340aac93e4801796f1c305cc071c82a80d1a2f2e575e34e6efde3957b4","observation_id":"11ce8e20-346e-4ee5-9948-e01920c03746","resolution":{"observed_at":"2026-05-16T09:46:10.073190Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2404.06395","last_updated":"2024-06-03T08:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-09T15:36:50Z","title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T18:00:53.389420Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2404.06395"},"observation_digest":"sha256:0ba49be00b6119baf1403c0084ef89ddd1bc327bb96b2310a67da129aa8a9d10","observation_id":"7ded08bb-9350-4ace-a73c-fa0057af11eb","resolution":{"observed_at":"2026-05-13T18:00:53.450722Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2501.05465","last_updated":"2026-05-14T16:52:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-03T19:53:57Z","title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T05:47:48.488826Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2501.05465"},"observation_digest":"sha256:415a89bc9a76e896493c6bdaa5326235c870ecdbd15f0634e959bb828f733074","observation_id":"7f2cbd0f-da97-422e-8f01-f1548ff192de","resolution":{"observed_at":"2026-05-23T05:52:37.423436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-08T13:44:00.474327Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.07832","last_updated":"2025-02-11T00:21:40Z","snapshot_observed_at":"2026-08-09T03:29:51.257494Z","submitted_at":"2025-02-11T00:21:40Z","title":"SHARP: Accelerating Language Model Inference by SHaring Adjacent layers with Recovery Parameters","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T13:44:00.474327Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2502.07832"},"observation_digest":"sha256:89e66fec098fcee3705a1b7d229f923f9cc437da8251aab83477c4728bc53bfe","observation_id":"8d69a650-a8bd-4bfe-bada-f1598e01d261","resolution":{"observed_at":"2026-08-08T13:44:00.474327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-07T14:44:23.116345Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17716","last_updated":"2025-05-23T10:33:14Z","snapshot_observed_at":"2026-08-08T15:42:18.320838Z","submitted_at":"2025-05-23T10:33:14Z","title":"Get Experience from Practice: LLM Agents with Record & Replay","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:44:23.116345Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2505.17716"},"observation_digest":"sha256:87275c964d4d9f424f9431d5244e3385c5cd18644d6e08a560ff21301617ff91","observation_id":"0d10b42b-515c-40a5-b47a-0fafba6430f3","resolution":{"observed_at":"2026-08-07T14:44:23.116345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-07T12:45:26.558466Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23725","last_updated":"2026-06-02T01:19:21Z","snapshot_observed_at":"2026-08-07T12:36:41.591936Z","submitted_at":"2025-05-29T17:55:37Z","title":"MuLoCo: Muon is a practical inner optimizer for DiLoCo","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:26.558466Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2505.23725"},"observation_digest":"sha256:d12e49f9daeab3b4b29aa95f446413fa2674b4f658c5de5ad71314035386707b","observation_id":"df1ccc47-c817-46e5-b64c-4ed0e368cc4e","resolution":{"observed_at":"2026-08-07T12:45:26.558466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-07T05:50:18.794719Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06926","last_updated":"2025-06-07T21:29:25Z","snapshot_observed_at":"2026-08-08T11:16:57.683716Z","submitted_at":"2025-06-07T21:29:25Z","title":"Basis Transformers for Multi-Task Tabular Regression","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:50:18.794719Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2506.06926"},"observation_digest":"sha256:6c44715d49b6f2204c97ed2a38484155dd254f2a6e6e784b478eb95baec2e3ba","observation_id":"7b19a67a-b969-4e9b-bd05-f3e374efaaa7","resolution":{"observed_at":"2026-08-07T05:50:18.794719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-06T11:44:04.928577Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.22448","last_updated":"2025-07-30T07:55:33Z","snapshot_observed_at":"2026-08-06T11:43:58.532083Z","submitted_at":"2025-07-30T07:55:33Z","title":"Falcon-H1: A Family of Hybrid-Head Language Models Redefining Efficiency and Performance","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T11:44:04.928577Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2507.22448"},"observation_digest":"sha256:f1f8a8abc1f2c3bae01b478d76f206103a821c35fe535e6800f69e5f0ab60b71","observation_id":"4c52a78a-a6e0-4ad8-a59b-a3e36ad36ad0","resolution":{"observed_at":"2026-08-06T11:44:04.928577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2509.18218","last_updated":"2026-04-06T17:04:13Z","snapshot_observed_at":"2026-07-06T22:30:31.933240Z","submitted_at":"2025-09-21T22:34:00Z","title":"Similarity Field Theory: A Mathematical Framework for Intelligence","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-18T14:36:49.543497Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2509.18218"},"observation_digest":"sha256:a958bab6e7ab9edf6caf7ed7242aaca3849910ec420f4ddbb7a2fd18fcd8f811","observation_id":"696b93b9-96f5-49dc-af04-c4a478f741f8","resolution":{"observed_at":"2026-05-18T14:41:30.649201Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2511.09447","last_updated":"2026-04-27T09:15:08Z","snapshot_observed_at":"2026-07-06T22:35:37.039024Z","submitted_at":"2025-11-12T16:03:52Z","title":"SpaDA: A Spatial Dataflow Architecture Programming Language","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T22:18:38.464461Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2511.09447"},"observation_digest":"sha256:6d7d6abcf8fb851256724fb2c65f9f45fec26ebd45476795ba840036e5822a9c","observation_id":"f6d2d069-c6f6-451d-9f9d-23f49452f5d0","resolution":{"observed_at":"2026-05-17T22:20:23.248975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2603.00541","last_updated":"2026-05-11T13:53:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-28T08:38:50Z","title":"Spectral Condition for $\\mu$P under Width-Depth Scaling","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T18:03:31.202134Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2603.00541"},"observation_digest":"sha256:7fc72df806720106f6e7d5ca3d372297b3a7421cf91510e161c51020a744cfd2","observation_id":"1c9812f9-a439-4926-ad6d-d58639597ea3","resolution":{"observed_at":"2026-05-15T18:06:25.514696Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2604.03199","last_updated":"2026-04-03T17:17:51Z","snapshot_observed_at":"2026-08-02T10:19:39.008617Z","submitted_at":"2026-04-03T17:17:51Z","title":"Learning the Signature of Memorization in Autoregressive Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T19:53:10.396785Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2604.03199"},"observation_digest":"sha256:967982415ea852546c3179980ced09d5485f3ed17180ba8aec1823b4ac3fccad","observation_id":"3030597b-90d3-49d4-b542-8bcf4d65b7e0","resolution":{"observed_at":"2026-05-13T19:53:11.472570Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2604.04722","last_updated":"2026-04-06T14:45:49Z","snapshot_observed_at":"2026-07-06T22:53:37.357996Z","submitted_at":"2026-04-06T14:45:49Z","title":"Don't Waste Bits! Adaptive KV-Cache Quantization for Lightweight On-Device LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T19:31:17.420639Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2604.04722"},"observation_digest":"sha256:9372fac02e6df2e5a5f86489bf72861cfa15b38c2819c75d88a801bef022718a","observation_id":"397fe58b-91bd-4d8c-832d-adabb3ac28a1","resolution":{"observed_at":"2026-05-10T22:55:48.072540Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2604.13275","last_updated":"2026-04-14T20:12:05Z","snapshot_observed_at":"2026-07-06T23:01:19.191464Z","submitted_at":"2026-04-14T20:12:05Z","title":"Better and Worse with Scale: How Contextual Entrainment Diverges with Model Size","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T15:18:33.458143Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2604.13275"},"observation_digest":"sha256:e5cdd481a11152660e6dc25210311c7f7ecd842fa78b854e3a675bb6e70c7032","observation_id":"a5243836-dd44-4be2-b867-29d28b90e009","resolution":{"observed_at":"2026-05-11T10:46:06.499168Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2605.15290","last_updated":"2026-05-14T18:03:16Z","snapshot_observed_at":"2026-07-06T23:26:37.362117Z","submitted_at":"2026-05-14T18:03:16Z","title":"GQA-{\\mu}P: The maximal parameterization update for grouped query attention","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-19T16:35:40.231293Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.15290"},"observation_digest":"sha256:5b4892f2fc9e6b03cefb78fcfbfbad4d4125c6c0159f3774c43dad7ba674a4e4","observation_id":"e53f190d-6c90-453b-8425-d49eb776e48f","resolution":{"observed_at":"2026-05-19T16:37:39.819716Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2605.15413","last_updated":"2026-05-14T20:57:15Z","snapshot_observed_at":"2026-08-08T21:55:40.844362Z","submitted_at":"2026-05-14T20:57:15Z","title":"Transformer Scalability Crisis: The First Comprehensive Empirical Analysis of Performance Walls in Modern Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-19T16:10:52.582492Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.15413"},"observation_digest":"sha256:a90311fbf2aa5d3d46068dde01c41ca188ef3ceb84c0e5b2c514804be259bc73","observation_id":"33fd6283-7502-4d89-822b-da9af1e96e67","resolution":{"observed_at":"2026-05-19T16:12:38.900193Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2605.29448","last_updated":"2026-07-18T01:59:39Z","snapshot_observed_at":"2026-08-06T23:52:37.250294Z","submitted_at":"2026-05-28T06:40:29Z","title":"How Much Is a Dataset Worth? Scaling Laws, the Vendi Score, and Matrix Spectral Functions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T08:33:05.952601Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.29448"},"observation_digest":"sha256:bfd661f29e5a321f8d89c167e380bc4736212cdebc2663de59bcb7b6c8963c97","observation_id":"e9e163fd-e946-4b77-a4fc-472196530f2d","resolution":{"observed_at":"2026-06-29T08:33:14.808607Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-02T12:58:43.750585Z","title":"et al.Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.29448","last_updated":"2026-07-18T01:59:39Z","snapshot_observed_at":"2026-08-06T23:52:37.250294Z","submitted_at":"2026-05-28T06:40:29Z","title":"How Much Is a Dataset Worth? Scaling Laws, the Vendi Score, and Matrix Spectral Functions","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T12:58:43.750585Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.29448"},"observation_digest":"sha256:2d88599a274c659567d958e847a96badccccb1ddcc2a8940020efc58f2461867","observation_id":"fd334a5c-971e-46d0-917f-5e35d581e4cb","resolution":{"observed_at":"2026-08-02T12:58:43.750585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.05029","last_updated":"2026-06-03T15:57:42Z","snapshot_observed_at":"2026-08-01T16:51:47.025025Z","submitted_at":"2026-06-03T15:57:42Z","title":"Validity Threats for Foundation Model Research","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T06:52:41.653304Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.05029"},"observation_digest":"sha256:81d32c92f03e71e13529dc9038c18eb8ed69ded8a71f04e252caf14513652026","observation_id":"e4d79cee-fbab-475c-a54c-8e6d81140666","resolution":{"observed_at":"2026-07-02T07:36:44.918374Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.05610","last_updated":"2026-06-04T02:32:11Z","snapshot_observed_at":"2026-07-06T23:45:33.157370Z","submitted_at":"2026-06-04T02:32:11Z","title":"Predictable Scaling Laws of Optimal Hyperparameters for LLM Continued Pre-training","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-28T01:53:04.715108Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.05610"},"observation_digest":"sha256:8c7aa4f585d82b0ee1a87d3729815ae94522ee37754a676af1a086ac1c60dbd7","observation_id":"83636f29-a75c-4049-8d5e-065e442f67a7","resolution":{"observed_at":"2026-07-02T12:46:56.727189Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.20299","last_updated":"2026-07-01T14:03:37Z","snapshot_observed_at":"2026-07-06T23:55:29.991442Z","submitted_at":"2026-06-18T14:35:53Z","title":"Statistical Properties of Training & Generalization","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-06-26T15:35:51.654392Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.20299"},"observation_digest":"sha256:0cefa47e1c0ca61b779883924c266e5c52be4382c27f61ac3591fcc552c00a3a","observation_id":"345089cb-143d-4052-8b8a-99f84689ca1e","resolution":{"observed_at":"2026-06-26T15:39:33.213970Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.20299","last_updated":"2026-07-01T14:03:37Z","snapshot_observed_at":"2026-07-06T23:55:29.991442Z","submitted_at":"2026-06-18T14:35:53Z","title":"Statistical Properties of Training & Generalization","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-07-02T21:51:13.457071Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.20299"},"observation_digest":"sha256:e2624c5353b69a1b5401643b2a3cacac57e90d3c18d90626f99a9da62ba69a50","observation_id":"4b39796f-3374-4299-852e-1e98a5359572","resolution":{"observed_at":"2026-07-02T21:57:25.386062Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.22873","last_updated":"2026-06-25T18:44:01Z","snapshot_observed_at":"2026-08-02T23:29:21.699637Z","submitted_at":"2026-06-22T05:37:43Z","title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-26T09:19:50.623741Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.22873"},"observation_digest":"sha256:bb908c5f1281908df41387cdfb7c936a859ea8915bec06bb9a585b6793d7ba5b","observation_id":"be861fad-ea9a-493a-834d-189e4735834a","resolution":{"observed_at":"2026-07-04T09:59:45.114260Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.22873","last_updated":"2026-06-25T18:44:01Z","snapshot_observed_at":"2026-08-02T23:29:21.699637Z","submitted_at":"2026-06-22T05:37:43Z","title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-29T01:18:19.195007Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.22873"},"observation_digest":"sha256:33c5229075ef5eefae0d91f9946b1132b890d1df4d0ad7884dbb87ca5fa54124","observation_id":"7eccc6d1-5602-4853-ae63-1f8ab990bf7f","resolution":{"observed_at":"2026-07-01T18:55:59.707947Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.24077","last_updated":"2026-06-23T02:42:33Z","snapshot_observed_at":"2026-08-03T17:16:10.837184Z","submitted_at":"2026-06-23T02:42:33Z","title":"Sentence-Level Contextual Entrainment in Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-26T00:52:53.959215Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.24077"},"observation_digest":"sha256:f74421bf52731ca4dabd0f0fb199500a764ff9c3a4260c48a59ad2b2e9abee97","observation_id":"970f52ba-dd88-4742-ad29-5b29c5482ca2","resolution":{"observed_at":"2026-07-04T16:09:57.234685Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2304.03208/citation-record","integrity":"/paper/2304.03208/integrity","json":"/paper/2304.03208/citation-record.json","paper":"/paper/2304.03208"},"outbound":[],"paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 26 inbound Pith citation observations for arXiv:2304.03208."}