{"as_of":"2026-07-31T22:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:735fc573209cb73db102d24d57226e6d32c3d15dbd514c4a26cc59c3a981418c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-07-31T06:34:12.847434+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-01T06:19:17.851285Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T13:59:52.014628Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2505.13196","last_updated":"2026-06-09T19:32:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-19T14:51:40Z","title":"A Physics-Inspired Optimizer: Velocity Regularized Adam","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T14:34:37.468180Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2505.13196"},"observation_digest":"sha256:d735649c3f0be3c07f07d0cb49bc72ada065621df397737185dc7ce7361e5512","observation_id":"b5c34744-65d8-4e17-b670-37cf072e8632","resolution":{"observed_at":"2026-05-22T14:34:54.236532Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2506.14951","last_updated":"2026-05-08T15:01:24Z","snapshot_observed_at":"2026-07-06T21:43:54.651353Z","submitted_at":"2025-06-17T20:04:15Z","title":"Flat Channels to Infinity in Neural Loss Landscapes","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-19T08:56:05.147928Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2506.14951"},"observation_digest":"sha256:01d08ae584a3871a55f34daeb32903a3fd1aa7cb32cd505b6cadabf50171c491","observation_id":"b9109b34-c621-40c6-a836-0ff90d2b81fc","resolution":{"observed_at":"2026-05-19T08:57:13.180565Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2510.21588","last_updated":"2026-04-10T15:56:31Z","snapshot_observed_at":"2026-07-06T22:33:58.663508Z","submitted_at":"2025-10-24T15:54:25Z","title":"Contribution of task-irrelevant stimuli to drift of neural representations","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-18T05:15:32.122551Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2510.21588"},"observation_digest":"sha256:0e7d75a875370e7dc5646fe69a95128d83d1d7e0c1b77652d49920b9431137c0","observation_id":"f4603a85-6774-493e-932b-69e80de4147e","resolution":{"observed_at":"2026-05-18T05:15:54.252256Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2604.14108","last_updated":"2026-04-15T17:28:46Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:28:46Z","title":"Momentum Further Constrains Sharpness at the Edge of Stochastic Stability","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-10T13:12:58.210967Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2604.14108"},"observation_digest":"sha256:3fe993817bb8d5ae2a11adcffaec12eda33ed57fd53f4ef5422aacb63edfed7c","observation_id":"62266676-88cc-494c-aa57-60709ed26025","resolution":{"observed_at":"2026-05-10T13:20:26.343484Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2604.19740","last_updated":"2026-04-21T17:59:02Z","snapshot_observed_at":"2026-07-06T23:06:21.774788Z","submitted_at":"2026-04-21T17:59:02Z","title":"Generalization at the Edge of Stability","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-10T02:37:13.715607Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2604.19740"},"observation_digest":"sha256:0e99a7eacde51aaeb655188dd04b37955a3f092e1745a0d62928d35a2c83309a","observation_id":"da9489f5-9e3d-40c3-be9d-e73b69c5c6ae","resolution":{"observed_at":"2026-05-10T02:38:17.376010Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2604.20219","last_updated":"2026-06-23T12:20:06Z","snapshot_observed_at":"2026-07-06T23:06:44.624357Z","submitted_at":"2026-04-22T06:09:20Z","title":"Layer-wise Geometric Approximation Rates for Deep Networks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T01:25:40.482499Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2604.20219"},"observation_digest":"sha256:e349e77a8df918591afeda881c9ac768e0e90c141f6e8338690ec2f00cb2b7c0","observation_id":"97ff5d34-7b0b-4cec-84e3-fc0defb83d1b","resolution":{"observed_at":"2026-05-11T13:36:07.856659Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2604.21691","last_updated":"2026-04-23T13:58:12Z","snapshot_observed_at":"2026-07-06T23:08:14.631453Z","submitted_at":"2026-04-23T13:58:12Z","title":"There Will Be a Scientific Theory of Deep Learning","version":1},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-05-09T20:11:17.616190Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2604.21691"},"observation_digest":"sha256:b702ec9376ed09872742b1368b8969d8e67ea33215271d11f83267fd46edb972","observation_id":"09268dd0-8a5f-44a0-bfbc-63af3c71e4ec","resolution":{"observed_at":"2026-05-11T15:21:09.133215Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2604.22778","last_updated":"2026-04-03T08:58:53Z","snapshot_observed_at":"2026-07-06T23:09:05.682387Z","submitted_at":"2026-04-03T08:58:53Z","title":"The Spectral Lifecycle of Transformer Training: Transient Compression Waves, Persistent Spectral Gradients, and the Q/K--V Asymmetry","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T20:56:56.460027Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2604.22778"},"observation_digest":"sha256:7676cc1361eb9e31c9c4b7279640289f46952cefd3d355868aea9beab5429f07","observation_id":"fc28b148-6b29-4d01-823e-5207b8ee2902","resolution":{"observed_at":"2026-05-13T20:58:15.707347Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2605.07870","last_updated":"2026-05-21T16:56:03Z","snapshot_observed_at":"2026-07-06T23:20:11.042952Z","submitted_at":"2026-05-08T15:28:01Z","title":"Spectral Dynamics in Deep Networks: Feature Learning, Outlier Escape, and Learning Rate Transfer","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-11T03:02:52.833353Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2605.07870"},"observation_digest":"sha256:951cb4159badb0698f9c6acb4ad6e4f5503517524210083e26138082018f7662","observation_id":"5415084a-48c8-4b62-bfd0-0c45f3a1af32","resolution":{"observed_at":"2026-05-11T03:05:53.385062Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2605.07870","last_updated":"2026-05-21T16:56:03Z","snapshot_observed_at":"2026-07-06T23:20:11.042952Z","submitted_at":"2026-05-08T15:28:01Z","title":"Spectral Dynamics in Deep Networks: Feature Learning, Outlier Escape, and Learning Rate Transfer","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-22T10:25:54.649302Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2605.07870"},"observation_digest":"sha256:4410d1f4c188e4513eb37a081be3a121dc212d95e5830cc284ebfe84a7b41e7f","observation_id":"52ef7d8d-4af6-4aaf-bafe-c716d5a388ce","resolution":{"observed_at":"2026-05-22T10:26:24.090330Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2605.11181","last_updated":"2026-05-11T19:42:48Z","snapshot_observed_at":"2026-07-06T23:23:02.025664Z","submitted_at":"2026-05-11T19:42:48Z","title":"Muon is Not That Special: Random or Inverted Spectra Work Just as Well","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-13T04:16:48.332851Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2605.11181"},"observation_digest":"sha256:053ae7463af6f7c30276812f06ef5f37769349beb83e5cf19c5b46d8c86a4a18","observation_id":"0003ae66-7cfe-4074-b0a2-e86fa509b454","resolution":{"observed_at":"2026-05-13T04:17:13.710639Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2605.14345","last_updated":"2026-05-14T04:09:51Z","snapshot_observed_at":"2026-07-06T23:25:49.545418Z","submitted_at":"2026-05-14T04:09:51Z","title":"Convergence of difference inclusions via a diameter criterion","version":1},"reference_index":142,"source":"arxiv_source","source_observed_at":"2026-05-15T02:21:30.228735Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2605.14345"},"observation_digest":"sha256:38efe95e577ab8fdc2e4ed93eba4f87ff2d5fcf15e8d4921e0deca483288a52c","observation_id":"6f93dd4d-da9d-4684-896c-683e334c08f9","resolution":{"observed_at":"2026-05-15T02:23:31.712040Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2605.16622","last_updated":"2026-05-15T20:43:26Z","snapshot_observed_at":"2026-07-06T23:27:43.762904Z","submitted_at":"2026-05-15T20:43:26Z","title":"Does Weight Decay Enhance Training Stability?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-20T19:49:01.351717Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2605.16622"},"observation_digest":"sha256:f0cecad65d6cc9be7fd43174c199a69fb6886a8289913edf32467767ebab70f1","observation_id":"8beb7f2e-f307-4a9c-a031-4898b8d4f14d","resolution":{"observed_at":"2026-05-20T19:53:42.974508Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2605.19291","last_updated":"2026-05-19T03:10:33Z","snapshot_observed_at":"2026-07-06T23:30:02.207108Z","submitted_at":"2026-05-19T03:10:33Z","title":"Factor Augmented High-Dimensional SGD","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-20T03:31:04.529612Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2605.19291"},"observation_digest":"sha256:f153f7fe1ee744ab0bb4f4c0d46c0ae094d93a4ec1c599407233ff72df8e9517","observation_id":"70d76439-9e88-4f86-8abe-521162f662ea","resolution":{"observed_at":"2026-05-20T03:33:01.672110Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.04662","last_updated":"2026-06-03T09:40:30Z","snapshot_observed_at":"2026-07-06T23:44:42.700120Z","submitted_at":"2026-06-03T09:40:30Z","title":"Why Muon Outperforms Adam: A Curvature Perspective","version":1},"reference_index":133,"source":"arxiv_source","source_observed_at":"2026-06-28T07:04:21.012269Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.04662"},"observation_digest":"sha256:5d550a639017f5f9b58d2dad1f7cbc294f16226e5c69c94f6b1aae05892fb5a0","observation_id":"c556039a-76c0-4e68-a8d2-1196cd84b577","resolution":{"observed_at":"2026-07-02T07:06:44.929203Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.05326","last_updated":"2026-06-03T18:11:33Z","snapshot_observed_at":"2026-07-31T13:54:06.376528Z","submitted_at":"2026-06-03T18:11:33Z","title":"Gradient descent at the Edge of Stability: free energy model and kinetic description of the two-layer network","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-06-28T05:00:47.062637Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.05326"},"observation_digest":"sha256:f6ed053f5bad846beeec87f3bd53d90d0949feae5cb2afb49c7a2c903dafebb7","observation_id":"21f6cb1f-6336-4f55-adec-9b2f04f3573d","resolution":{"observed_at":"2026-07-02T10:36:52.024353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.09762","last_updated":"2026-06-08T17:24:15Z","snapshot_observed_at":"2026-07-06T23:49:03.237958Z","submitted_at":"2026-06-08T17:24:15Z","title":"Preserving Plasticity in Continual Learning via Dynamical Isometry","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T17:26:20.769515Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.09762"},"observation_digest":"sha256:d04d46190032052f0121c78ebe7e8ff4879ae014cc842fd11d24817b310562e6","observation_id":"3a27da96-2448-452a-b926-44eb2e35705c","resolution":{"observed_at":"2026-07-03T00:07:28.533864Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.12883","last_updated":"2026-06-11T04:19:32Z","snapshot_observed_at":"2026-07-06T23:51:45.306189Z","submitted_at":"2026-06-11T04:19:32Z","title":"The Hidden Power of Scaling Factor in LoRA Optimization","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-06-27T07:14:08.479610Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.12883"},"observation_digest":"sha256:c11eb917496b812b47967d015fe6081dadd8a80a0570f392751b1a47f4b3c863","observation_id":"24a244be-c2b4-48de-b867-027fbd1872cb","resolution":{"observed_at":"2026-07-03T14:08:21.893561Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.19179","last_updated":"2026-06-17T15:19:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-17T15:19:20Z","title":"Compute Efficiency and Serial Runtime Tradeoffs for Stochastic Momentum Methods","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T21:31:49.871520Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.19179"},"observation_digest":"sha256:8c455419b4b15ce6aa843f44151f50df1465ce0f5ea71de974771aa5287468ea","observation_id":"0afe8914-c7f2-4f60-8274-56f4bc685d12","resolution":{"observed_at":"2026-07-03T23:59:07.495055Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.19491","last_updated":"2026-06-17T18:28:37Z","snapshot_observed_at":"2026-07-31T17:54:27.925523Z","submitted_at":"2026-06-17T18:28:37Z","title":"Algebraic Dead Directions in LayerNorm Transformers: A Forward-Pass-Only Diagnostic at LLM Scale","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-26T21:14:11.815522Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.19491"},"observation_digest":"sha256:22693f601bb898104277577081a77abe6bba801409dc302eba6486dcda28ae0a","observation_id":"caf1d49f-8eb8-4e9b-be18-079834fa318c","resolution":{"observed_at":"2026-07-04T00:29:15.278476Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.21514","last_updated":"2026-06-19T15:10:20Z","snapshot_observed_at":"2026-07-06T23:56:31.454439Z","submitted_at":"2026-06-19T15:10:20Z","title":"Towards Understanding the Power and Limits of the Muon Optimizer: A River-Valley Perspective","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-26T14:22:54.988008Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.21514"},"observation_digest":"sha256:469f4e0e775d88f4a74a747c09e665332a099a936e221c2d886cd0960558dd9d","observation_id":"76d6279e-7039-4194-8d26-d0990a0a8214","resolution":{"observed_at":"2026-07-04T06:29:38.158808Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.26492","last_updated":"2026-06-25T00:52:22Z","snapshot_observed_at":"2026-07-31T03:59:24.023131Z","submitted_at":"2026-06-25T00:52:22Z","title":"Evaluation-Strategy Gap in Fault Diagnosis of Deep Learning Programs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-26T04:45:46.785605Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.26492"},"observation_digest":"sha256:ad63634bd5e50e5f2ef8c7f252cbd436448f2eb126999900502623655fba3466","observation_id":"3cdaa2a7-dee8-4957-810a-7b8432c70ef2","resolution":{"observed_at":"2026-07-04T13:59:52.016229Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","version":3},"cited_work":{"arxiv_id":"2103.00065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.00065","snapshot_observed_at":"2026-07-04T13:59:52.014628Z","title":"org/abs/2103.00065","venue":null,"work_id":"aac67922-3bad-4699-b070-cac34775f0c2","year":2021},"citing_paper":{"arxiv_id":"2606.32000","last_updated":"2026-06-30T17:34:13Z","snapshot_observed_at":"2026-07-07T00:05:37.068846Z","submitted_at":"2026-06-30T17:34:13Z","title":"Radial Suppression Accelerates Algorithmic Generalization: A Geometric Analysis of Delayed Generalization","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-07-01T06:19:17.851285Z"},"links":{"cited_paper":"/paper/2103.00065","citing_paper":"/paper/2606.32000"},"observation_digest":"sha256:1a8da6061cd344e0224a9fd02bfe122cec5a30608a2ce3135b1f05940e528f39","observation_id":"09a69c9f-9173-43b3-9a22-4a954a069fa4","resolution":{"observed_at":"2026-07-01T09:45:39.731226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-31T06:34:12.847434+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2103.00065/citation-record","integrity":"/paper/2103.00065/integrity","json":"/paper/2103.00065/citation-record.json","paper":"/paper/2103.00065"},"outbound":[],"paper":{"arxiv_id":"2103.00065","last_updated":"2022-11-23T18:09:57Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T10:45:06.227629Z","submitted_at":"2021-02-26T22:08:19Z","title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-07-31T06:34:12.847434+00:00","source":"crossref"},{"observed_at":"2026-07-31T06:34:08.642788+00:00","source":"retraction_watch"}],"thesis":"As of 31 July 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2103.00065."}