{"as_of":"2026-08-22T08:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a7ff1aa4e4c74218f55dcc0fbc8df5864457ad3050d14a08ac5597aa05a82ace","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T04:35:57.949886Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2409.20325","last_updated":"2024-12-06T14:09:22Z","snapshot_observed_at":"2026-08-12T12:26:45.359430Z","submitted_at":"2024-09-30T14:26:12Z","title":"Old Optimizer, New Norm: An Anthology","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-16T07:27:52.883335Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2409.20325"},"observation_digest":"sha256:e38c1e630b87e6b29a23312e24163787bf2c663fa09485bd00100a8fe2b06140","observation_id":"9215efe9-e270-4cbb-b857-68acc22d380f","resolution":{"observed_at":"2026-05-16T07:27:52.941479Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-12T19:43:01.988790Z","title":"M., Lee, T.-H., Iwasaki, S., Gallego-Posada, J., Li, Z., Rangadurai, K., Mudigere, D., and Rabbat, M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10438","last_updated":"2025-09-04T09:37:05Z","snapshot_observed_at":"2026-08-17T20:25:32.027243Z","submitted_at":"2024-11-15T18:57:39Z","title":"MARS: Unleashing the Power of Variance Reduction for Training Large Models","version":4},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-12T19:43:01.988790Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2411.10438"},"observation_digest":"sha256:c992258b9f9bc3a94d365894815123ab43b004ff1037e7137672009bc2741d4d","observation_id":"840b886c-4ca4-4976-9d25-026b41b288c8","resolution":{"observed_at":"2026-08-12T19:43:01.988790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-12T06:03:49.313548Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.19617","last_updated":"2024-11-29T11:10:29Z","snapshot_observed_at":"2026-08-17T17:57:37.030720Z","submitted_at":"2024-11-29T11:10:29Z","title":"Materials Learning Algorithms (MALA): Scalable Machine Learning for Electronic Structure Calculations in Large-Scale Atomistic Simulations","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T06:03:49.313548Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2411.19617"},"observation_digest":"sha256:9697981ce506a78897f6610fdefce5cfb206efa9398f03b84dd12d485cd07d79","observation_id":"1458026b-f75b-4f94-84e3-c5f9a354b5c9","resolution":{"observed_at":"2026-08-12T06:03:49.313548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-11T15:50:24.267978Z","title":"A distributed data-parallel pytorch implementation of the distributed shampoo optimizer for training neural networks at-scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10663","last_updated":"2025-03-12T09:19:31Z","snapshot_observed_at":"2026-08-16T18:13:45.373481Z","submitted_at":"2024-12-14T03:32:54Z","title":"Memory-Efficient 4-bit Preconditioned Stochastic Optimization","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T15:50:24.267978Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2412.10663"},"observation_digest":"sha256:b736d94b15c23c183fbb125f49d8dbe7bf78d2463fa5b593db7a087f3f06fb21","observation_id":"4da20aa6-8ab7-4c1e-8e21-8c47e36fb61a","resolution":{"observed_at":"2026-08-11T15:50:24.267978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-11T13:28:29.142246Z","title":"A distributed data-parallel pytorch implementation of the distributed shampoo optimizer for training neural networks at-scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13148","last_updated":"2025-02-21T18:59:37Z","snapshot_observed_at":"2026-08-15T15:12:45.559733Z","submitted_at":"2024-12-17T18:13:18Z","title":"SWAN: SGD with Normalization and Whitening Enables Stateless LLM Training","version":3},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-11T13:28:29.142246Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2412.13148"},"observation_digest":"sha256:17e99d9db0d82409b9f071677b15acc12fdc3c468db28ebcc27af54c7e718c53","observation_id":"50362495-f249-4a7d-bd41-0e1ed9c1767d","resolution":{"observed_at":"2026-08-11T13:28:29.142246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-09T14:47:40.724257Z","title":"M., Lee, T.-H., Iwasaki, S., Gallego-Posada, J., Li, Z., Rangadurai, K., Mudigere, D., and Rabbat, M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01763","last_updated":"2025-02-03T19:08:32Z","snapshot_observed_at":"2026-08-18T05:28:06.282520Z","submitted_at":"2025-02-03T19:08:32Z","title":"On The Concurrence of Layer-wise Preconditioning Methods and Provable Feature Learning","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-09T14:47:40.724257Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2502.01763"},"observation_digest":"sha256:2b1a821a7e94aec6847c0acb34d2158c02f53b81c78d6be7e140405cd64c4037","observation_id":"5ce83e7f-3c9c-4240-9c0a-ed5c29df1bf1","resolution":{"observed_at":"2026-08-09T14:47:40.724257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-08T16:15:07.336710Z","title":"M., Lee, T.-H., Iwasaki, S., Gallego-Posada, J., Li, Z., Rangadurai, K., Mudigere, D., and Rabbat, M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06268","last_updated":"2025-03-28T15:49:41Z","snapshot_observed_at":"2026-08-15T11:58:50.867462Z","submitted_at":"2025-02-10T09:07:04Z","title":"Spectral-factorized Positive-definite Curvature Learning for NN Training","version":3},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-08T16:15:07.336710Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2502.06268"},"observation_digest":"sha256:e39071b8641a2d003f01f9e43814543533d9816b907f6ff726a591a9d07ca205","observation_id":"a5ee00b1-8cf5-4761-bc55-477f7dd8b99e","resolution":{"observed_at":"2026-08-08T16:15:07.336710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-16T04:35:57.949886Z","title":"A distributed data-parallel pytorch implementation of the distributed shampoo optimizer for training neural networks at-scale,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00982","last_updated":"2025-08-04T15:48:47Z","snapshot_observed_at":"2026-08-19T15:13:12.987961Z","submitted_at":"2025-05-02T04:02:36Z","title":"DHO$_2$: Accelerating Distributed Hybrid Order Optimization via Model Parallelism and ADMM","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T04:35:57.949886Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2505.00982"},"observation_digest":"sha256:f8a4e15bd49e4449676c66d566add909e910e7c2fa4d6ecff71f3ad7981e88cf","observation_id":"9b2fc8e5-cbe0-4a3f-9627-e9adc83f1ef1","resolution":{"observed_at":"2026-08-16T04:35:57.949886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-15T20:18:22.869640Z","title":"A distributed data-parallel pytorch implementation of the distributed shampoo optimizer for training neural networks at-scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13397","last_updated":"2025-05-19T17:34:32Z","snapshot_observed_at":"2026-08-20T05:17:37.987687Z","submitted_at":"2025-05-19T17:34:32Z","title":"Learning by solving differential equations","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T20:18:22.869640Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2505.13397"},"observation_digest":"sha256:d0339110be758c1d364911ff0fd7162705d5911eeedc439e45c4d34d659b74cb","observation_id":"774ec820-ea2a-42c4-9cb4-74490d2ff5f8","resolution":{"observed_at":"2026-08-15T20:18:22.869640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T11:03:40.394757Z","title":"A distributed data-parallel P y T orch implementation of the distributed S hampoo optimizer for training neural networks at-scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03378","last_updated":"2026-06-22T12:44:13Z","snapshot_observed_at":"2026-08-14T02:06:11.919893Z","submitted_at":"2025-09-03T14:55:15Z","title":"Understanding and Improving Shampoo and SOAP via Kullback-Leibler Minimization","version":10},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T11:03:40.394757Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2509.03378"},"observation_digest":"sha256:366138e3da77931b27640fbb6052d0796a8decd0b342d4da28a1759fb3e4ebe6","observation_id":"a2ee34b3-b56c-4a37-a986-75b2883a309f","resolution":{"observed_at":"2026-08-05T11:03:40.394757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2603.20527","last_updated":"2026-05-13T10:41:39Z","snapshot_observed_at":"2026-08-14T09:30:22.665664Z","submitted_at":"2026-03-20T21:55:28Z","title":"RMNP: Row-Momentum Normalized Preconditioning for Scalable Matrix-Based Optimization","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T07:49:39.417566Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2603.20527"},"observation_digest":"sha256:ea32153781c4a1cb7161945730cbaa5ad9a5045455d8b26206097526e6373187","observation_id":"9ec66fb5-b447-420e-8fef-c969e22a2760","resolution":{"observed_at":"2026-05-15T07:49:50.787770Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2605.04418","last_updated":"2026-05-06T02:22:06Z","snapshot_observed_at":"2026-08-15T03:40:21.944750Z","submitted_at":"2026-05-06T02:22:06Z","title":"Demystifying Manifold Constraints in LLM Pre-training","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-08T17:44:44.438637Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2605.04418"},"observation_digest":"sha256:c78fffbeb6794dab92f41df1f5127eab6b422f61b474a2fc55f0019c919d4e66","observation_id":"f7799bca-5bf9-4092-bb26-3a19778cd3b2","resolution":{"observed_at":"2026-05-11T17:16:08.604600Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2605.06316","last_updated":"2026-05-07T14:16:16Z","snapshot_observed_at":"2026-08-13T04:47:11.433459Z","submitted_at":"2026-05-07T14:16:16Z","title":"Pro-KLShampoo: Projected KL-Shampoo with Whitening Recovered by Orthogonalization","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-08T13:06:05.234216Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2605.06316"},"observation_digest":"sha256:5999d910a0370de3bb25419952f4ad7d9dc9dcf11d761e06b34bc7d0b33ed1f6","observation_id":"de53d088-01a2-481a-be0b-cd623329dde2","resolution":{"observed_at":"2026-05-11T19:01:12.783437Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2605.09552","last_updated":"2026-05-10T14:11:22Z","snapshot_observed_at":"2026-08-16T09:16:25.656625Z","submitted_at":"2026-05-10T14:11:22Z","title":"Phases of Muon: When Muon Eclipses SignSGD","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-12T04:05:08.899876Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2605.09552"},"observation_digest":"sha256:5ca263b9f5ac93553eb905542e06346c58bf53f925fa36042d1698a1687a7d42","observation_id":"29a8efb0-f48f-454f-8e0d-08b8ac786a8f","resolution":{"observed_at":"2026-05-12T06:41:29.878617Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2605.16165","last_updated":"2026-05-15T16:45:56Z","snapshot_observed_at":"2026-08-12T21:47:55.560127Z","submitted_at":"2026-05-15T16:45:56Z","title":"Second-Order Multi-Level Variance Correction for Modality Competition in Multimodal Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-20T19:02:02.266052Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2605.16165"},"observation_digest":"sha256:de47dc7964e1a2ddff972f0eda3c173084f708c96497f2e8c166468a09b19690","observation_id":"5e8147c0-a478-4d4c-a0a2-7757f8f6de36","resolution":{"observed_at":"2026-05-20T19:03:39.552743Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2605.16184","last_updated":"2026-05-15T17:03:55Z","snapshot_observed_at":"2026-08-22T06:06:11.034591Z","submitted_at":"2026-05-15T17:03:55Z","title":"Runtime-Orchestrated Second-Order Optimization for Scalable LLM Training","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T18:30:24.657592Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2605.16184"},"observation_digest":"sha256:ff4c12f63d406a475ba054392697ad83c30b41f8b449cb5eeb050f531a3205fc","observation_id":"27b85356-b987-4a75-ae0a-e606698c19aa","resolution":{"observed_at":"2026-05-19T18:32:42.896458Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2605.18106","last_updated":"2026-06-22T06:06:33Z","snapshot_observed_at":"2026-08-14T08:05:42.459480Z","submitted_at":"2026-05-18T09:17:26Z","title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","version":1},"reference_index":139,"source":"pdf_text","source_observed_at":"2026-05-20T09:34:45.186929Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2605.18106"},"observation_digest":"sha256:c9557f4fae51ca277498b4c48d8f09b5501da857c082ed4c0dc0f6c9c76b1405","observation_id":"2e523952-ef52-4d32-9045-51b22cf1a7e0","resolution":{"observed_at":"2026-05-20T09:38:11.084495Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2605.18106","last_updated":"2026-06-22T06:06:33Z","snapshot_observed_at":"2026-08-14T08:05:42.459480Z","submitted_at":"2026-05-18T09:17:26Z","title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","version":4},"reference_index":141,"source":"pdf_text","source_observed_at":"2026-06-30T18:42:01.854481Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2605.18106"},"observation_digest":"sha256:922167367c0b31e85492179b09db83c91dbbc07be6c635b0469fa2f84ee67c11","observation_id":"8899af77-1701-440b-9ed2-a0f582976f4a","resolution":{"observed_at":"2026-06-30T18:45:00.341379Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2606.02365","last_updated":"2026-06-01T15:13:28Z","snapshot_observed_at":"2026-08-12T17:48:57.483330Z","submitted_at":"2026-06-01T15:13:28Z","title":"FOAM: Frequency and Operator Error-Based Adaptive Damping Method for Reducing Staleness-Oriented Error for Shampoo","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-06-28T15:49:18.685160Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2606.02365"},"observation_digest":"sha256:68a05f310f8c35fcc33961e17aa64a3d7760ffdfdbd5ae29c00fdbd03a3a9629","observation_id":"23a46ad9-1f96-4330-8661-9f5a74ec8462","resolution":{"observed_at":"2026-07-01T22:06:16.001137Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2606.04058","last_updated":"2026-06-05T12:50:11Z","snapshot_observed_at":"2026-08-06T02:55:51.608023Z","submitted_at":"2026-06-02T11:31:21Z","title":"Spectral Scaling Laws of Muon","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T11:19:05.939340Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2606.04058"},"observation_digest":"sha256:6f1b8c1530cf18323e038700bbdaf224f6ac444e4c91b148266c6a51bee41842","observation_id":"0b05e4df-5e09-4577-b05a-e3ff329b9e69","resolution":{"observed_at":"2026-07-02T02:06:26.379608Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2606.06418","last_updated":"2026-06-04T17:22:58Z","snapshot_observed_at":"2026-08-16T10:17:57.703806Z","submitted_at":"2026-06-04T17:22:58Z","title":"Double Preconditioning (DoPr): Optimization for Test-Time Performance, not Validation Loss","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-06-28T02:35:39.845487Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2606.06418"},"observation_digest":"sha256:d7a7ce82d6dc5bbd08c68c7f7dabd60ddd74c70c5877c6ea3dcda116f717651c","observation_id":"14c2112f-286a-498c-ab05-649dc17b70d9","resolution":{"observed_at":"2026-07-02T12:06:55.400219Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2606.23357","last_updated":"2026-07-09T05:20:17Z","snapshot_observed_at":"2026-08-08T21:08:12.419631Z","submitted_at":"2026-06-22T13:56:59Z","title":"SOAP-Bubbles: Structured Weight Uncertainty for Neural Networks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T09:00:50.315383Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2606.23357"},"observation_digest":"sha256:fe7e544c509e3fbca0fa1fb239b7d728c6ee87d17208c3915d3358fedbe10621","observation_id":"0f602b61-e958-4214-81c3-0332ca7b31a6","resolution":{"observed_at":"2026-07-04T10:19:47.219672Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2606.27153","last_updated":"2026-06-25T15:23:03Z","snapshot_observed_at":"2026-08-12T12:41:24.191413Z","submitted_at":"2026-06-25T15:23:03Z","title":"DMuon: Efficient Distributed Muon Training with Near-Adam Overhead","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-26T02:41:02.917064Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2606.27153"},"observation_digest":"sha256:cf348e4ca5771b650ac3f1517775f47ffa9c36727e3710956f71830fe8ea30f0","observation_id":"d4ed3575-bfbf-49d8-86df-f929d40420ba","resolution":{"observed_at":"2026-07-04T14:39:58.558807Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":"2309.06497","doi":"10.48550/arxiv.2309.06497","metadata_source":"pith","pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2309.06497 (2023)","venue":"cs.LG","work_id":"12f3e87b-1374-4a12-943c-42087610a1a6","year":2023},"citing_paper":{"arxiv_id":"2607.05895","last_updated":"2026-07-07T06:45:39Z","snapshot_observed_at":"2026-08-12T14:26:31.779365Z","submitted_at":"2026-07-07T06:45:39Z","title":"MatrixFSDP: communication-free matrix optimizers under ZeRO-3 parameter sharding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-08T21:33:20.705837Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2607.05895"},"observation_digest":"sha256:9350b5af1271eb4851c9349e845fcf2286fcd1702b0d9873945cedc8793babce","observation_id":"8afea9ee-df30-476b-8d85-8941f75027c9","resolution":{"observed_at":"2026-07-08T21:35:37.643914Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-02T01:58:57.470749Z","title":"A distributed data-parallel pytorch imple- mentation of the distributed shampoo optimizer for training neural networks at-scale.arXiv preprint arXiv:2309.06497,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14536","last_updated":"2026-07-16T03:42:21Z","snapshot_observed_at":"2026-08-14T00:59:23.434969Z","submitted_at":"2026-07-16T03:42:21Z","title":"Muse: Representation Geometry of Muon Beyond Normalized Momentum","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T01:58:57.470749Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2607.14536"},"observation_digest":"sha256:7fa368c2acbdec2e7a1dcceabe153a80dc1cfae2305e1a20599656af4ff91573","observation_id":"6d826c50-3951-4413-bbf3-3c7d8feafd5f","resolution":{"observed_at":"2026-08-02T01:58:57.470749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-01T17:32:35.758665Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks at-Scale","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17620","last_updated":"2026-07-20T07:16:31Z","snapshot_observed_at":"2026-08-14T17:02:17.072637Z","submitted_at":"2026-07-20T07:16:31Z","title":"PoLoRA: A Preconditioned Orthogonalized LoRA Optimizer","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-01T17:32:35.758665Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2607.17620"},"observation_digest":"sha256:60eddf2bfbb36a5cd262422bdc774f5060380bb4d7344e0b03f99e3121d4d379","observation_id":"891d468f-8722-410b-92a1-9391048fd58d","resolution":{"observed_at":"2026-08-01T17:32:35.758665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-01T07:21:46.180322Z","title":"2309.06497 , archivePrefix=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27475","last_updated":"2026-07-31T05:10:39Z","snapshot_observed_at":"2026-08-19T09:52:42.343090Z","submitted_at":"2026-07-29T21:29:00Z","title":"OneShot: Index-in-Ranking with Neural Scoring for Large-Scale Retrieval","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-01T07:21:46.180322Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2607.27475"},"observation_digest":"sha256:3b5718b503b348347b069e7c3cb852cfaae7aeb3792f6fd8316a0b5286a9c826","observation_id":"fc89e38e-cb7b-4ec1-99c0-eda052520eaf","resolution":{"observed_at":"2026-08-01T07:21:46.180322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-03T01:45:23.261226Z","title":"2309.06497 , archivePrefix=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27475","last_updated":"2026-07-31T05:10:39Z","snapshot_observed_at":"2026-08-19T09:52:42.343090Z","submitted_at":"2026-07-29T21:29:00Z","title":"OneShot: Index-in-Ranking with Neural Scoring for Large-Scale Retrieval","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-03T01:45:23.261226Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2607.27475"},"observation_digest":"sha256:9880cc8a2a4cc1ec99b7e7a355ce452fc9ef73a12ddc4d50a2f1320c7af5b082","observation_id":"18859707-d211-499f-8c3e-d8fcb74962e9","resolution":{"observed_at":"2026-08-03T01:45:23.261226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-05T05:27:17.313102Z","title":"A distributed data-parallel pytorch implementation of the distributed shampoo optimizer for training neural networks at-scale,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.03941","last_updated":"2026-08-04T17:10:47Z","snapshot_observed_at":"2026-08-20T03:50:45.906112Z","submitted_at":"2026-08-04T17:10:47Z","title":"Muon Meets Mamba: Spectral Optimization for State Space Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T05:27:17.313102Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2608.03941"},"observation_digest":"sha256:87472ef178f5890270d0077a60c04a8862e9bec186f95317f1721ac67c85e35c","observation_id":"6ec7bf56-7c37-4897-b8e9-256fc8d7d013","resolution":{"observed_at":"2026-08-05T05:27:17.313102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06497","snapshot_observed_at":"2026-08-11T11:16:02.781870Z","title":"arXiv preprint arXiv:2309.06497 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09763","last_updated":"2026-08-10T15:55:07Z","snapshot_observed_at":"2026-08-19T01:55:21.619857Z","submitted_at":"2026-08-10T15:55:07Z","title":"Second-Order Muon Done Right: A Principled Marriage of Spectral Geometry and Curvature","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T11:16:02.781870Z"},"links":{"cited_paper":"/paper/2309.06497","citing_paper":"/paper/2608.09763"},"observation_digest":"sha256:f486ca4e1792a56e6f75e2327c8e8eb2c1ddf7e12980a6895537af7b5ba3f502","observation_id":"4a0b3809-ca41-41cc-aa65-bcf638e00f14","resolution":{"observed_at":"2026-08-11T11:16:02.781870Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2309.06497/citation-record","integrity":"/paper/2309.06497/integrity","json":"/paper/2309.06497/citation-record.json","paper":"/paper/2309.06497"},"outbound":[],"paper":{"arxiv_id":"2309.06497","last_updated":"2023-09-12T18:11:10Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-19T05:42:54.006296Z","submitted_at":"2023-09-12T18:11:10Z","title":"A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2309.06497."}