{"as_of":"2026-08-06T01:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d01cf4644d712e1df3ad754bbeda089ac97665a9498062bc74d7ee7784a8e38c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":11,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":11,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-01T00:37:16.364388Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":22,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2401.01335","last_updated":"2024-06-14T21:17:17Z","snapshot_observed_at":"2026-07-06T17:10:56.398607Z","submitted_at":"2024-01-02T18:53:13Z","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","version":3},"reference_index":160,"source":"arxiv_source","source_observed_at":"2026-05-14T23:00:20.720030Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2401.01335"},"observation_digest":"sha256:454825ac437b0d132263dd23b1509c6dc0daf32101e59c49de1ce3f72d748c82","observation_id":"9fbe3cd9-3aca-4a2f-bd38-aad8767b1ed4","resolution":{"observed_at":"2026-05-14T23:00:21.450892Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2511.09290","last_updated":"2026-05-05T12:11:24Z","snapshot_observed_at":"2026-07-06T22:35:37.039024Z","submitted_at":"2025-11-12T12:57:19Z","title":"Prediction horizon shapes representations in predictive learning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T23:29:11.660779Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2511.09290"},"observation_digest":"sha256:5150588542cf47c1828bb546cf9b294d082f503ec1dc83683ad4cf6f71182fb9","observation_id":"c1c9018d-85d8-4613-b9ef-a9b1cac8049a","resolution":{"observed_at":"2026-05-17T23:30:29.285468Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2602.01642","last_updated":"2026-05-07T23:28:50Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T04:59:24Z","title":"The Effect of Mini-Batch Noise on the Implicit Bias of Adam","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T08:03:15.785558Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2602.01642"},"observation_digest":"sha256:b186c5e7b4f8ec90883231ace2700d2fb3933c3b2a8fcc368de0991f860e5d74","observation_id":"64d2a43b-906c-4817-870c-e6afb88a40a6","resolution":{"observed_at":"2026-05-16T08:07:34.293910Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2603.02622","last_updated":"2026-04-10T00:16:23Z","snapshot_observed_at":"2026-08-02T08:31:17.613147Z","submitted_at":"2026-03-03T05:49:24Z","title":"Implicit Bias in Deep Linear Discriminant Analysis","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T16:48:21.791575Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2603.02622"},"observation_digest":"sha256:95ed03a3f9d7f6798ffee9001f753e7f87b41eaa4aca23378a5bd0abeaa18400","observation_id":"5c7d5a50-b564-460d-9995-6094367d7ea9","resolution":{"observed_at":"2026-05-15T16:50:10.520373Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2605.01288","last_updated":"2026-06-23T15:11:40Z","snapshot_observed_at":"2026-07-06T23:14:33.653260Z","submitted_at":"2026-05-02T06:55:15Z","title":"A Theory of Saddle Escape in Deep Nonlinear Networks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-09T14:54:47.763122Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2605.01288"},"observation_digest":"sha256:7d92370433d5805649ee2f8c31cf4df4b8937e225f76fe3b72e2af6cb6a2a70a","observation_id":"08ea6666-a977-4098-b3b0-ae34e07774dd","resolution":{"observed_at":"2026-05-11T16:51:05.759084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2605.01288","last_updated":"2026-06-23T15:11:40Z","snapshot_observed_at":"2026-07-06T23:14:33.653260Z","submitted_at":"2026-05-02T06:55:15Z","title":"A Theory of Saddle Escape in Deep Nonlinear Networks","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-11T02:22:38.751375Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2605.01288"},"observation_digest":"sha256:bbf024853d7c2fc2c9808fc493a2662a35f950e561809711654b13c007867e89","observation_id":"4982cdbd-8c6c-454c-9722-b0abd0166899","resolution":{"observed_at":"2026-05-11T02:25:54.529769Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2605.01288","last_updated":"2026-06-23T15:11:40Z","snapshot_observed_at":"2026-07-06T23:14:33.653260Z","submitted_at":"2026-05-02T06:55:15Z","title":"A Theory of Saddle Escape in Deep Nonlinear Networks","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-01T00:37:16.364388Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2605.01288"},"observation_digest":"sha256:cb19cc1acb7fc60252764c169ae7b2de721ef3cb5b6d1bffe910a09f8b02b24e","observation_id":"b536fac9-d327-4687-b69f-0e632e4ce183","resolution":{"observed_at":"2026-07-01T00:45:12.078658Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2605.10775","last_updated":"2026-05-11T16:08:23Z","snapshot_observed_at":"2026-08-03T08:28:30.711991Z","submitted_at":"2026-05-11T16:08:23Z","title":"On the global convergence of gradient descent for wide shallow models with bounded nonlinearities","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-12T03:51:08.871267Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2605.10775"},"observation_digest":"sha256:26398c827d96a165352663551210f3a1a1d3c92ccba875e279466a7a93580517","observation_id":"7cd8703b-570e-43db-9de6-244da623ccf6","resolution":{"observed_at":"2026-05-12T06:51:31.351983Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2606.10089","last_updated":"2026-06-08T19:16:32Z","snapshot_observed_at":"2026-08-02T05:36:24.589418Z","submitted_at":"2026-06-08T19:16:32Z","title":"A Theory on Flow Matching with Neural Networks","version":1},"reference_index":186,"source":"arxiv_source","source_observed_at":"2026-06-27T16:59:34.084575Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2606.10089"},"observation_digest":"sha256:9974125ff45182142c1c856cfd72c2e4f11d7eb96e82bd0d595f69d05171baa9","observation_id":"d0003375-3f8c-426a-b74b-19ee20a8b0ff","resolution":{"observed_at":"2026-07-03T00:47:30.982898Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2606.10913","last_updated":"2026-06-09T14:23:34Z","snapshot_observed_at":"2026-07-06T23:50:01.838569Z","submitted_at":"2026-06-09T14:23:34Z","title":"Conservation Laws from Data Symmetry in Neural Networks","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-27T13:53:48.656780Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2606.10913"},"observation_digest":"sha256:784fe740c490a12168556d7c69c469c7641f8c5ac7c522c9a0c74d68933809ca","observation_id":"548f2c45-f15c-4253-b8d8-7abf240ca5c8","resolution":{"observed_at":"2026-07-03T04:27:36.920204Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks","version":2},"cited_work":{"arxiv_id":"1810.02032","doi":"10.48550/arxiv.1810.02032","metadata_source":"pith","pith_arxiv_id":"1810.02032","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gradient descent aligns the layers of deep linear networks","venue":"cs.LG","work_id":"c044b7b0-50a1-4afe-87f7-9a88cb62fb5f","year":2018},"citing_paper":{"arxiv_id":"2606.17816","last_updated":"2026-06-16T11:44:53Z","snapshot_observed_at":"2026-08-05T19:42:53.795062Z","submitted_at":"2026-06-16T11:44:53Z","title":"Conservation Laws for Modern Neural Architectures","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-27T01:43:45.287360Z"},"links":{"cited_paper":"/paper/1810.02032","citing_paper":"/paper/2606.17816"},"observation_digest":"sha256:687ac1cf85a9e2a7d0ee51d7fc4aad366ee3a3562aaa7fc023242dc9dac55499","observation_id":"f12b4763-c9a2-4f83-875a-f89169785ca5","resolution":{"observed_at":"2026-07-03T19:58:54.968938Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/1810.02032/citation-record","integrity":"/paper/1810.02032/integrity","json":"/paper/1810.02032/citation-record.json","paper":"/paper/1810.02032"},"outbound":[],"paper":{"arxiv_id":"1810.02032","last_updated":"2019-02-24T10:28:05Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-05T20:58:00.508784Z","submitted_at":"2018-10-04T02:48:41Z","title":"Gradient descent aligns the layers of deep linear networks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 11 inbound Pith citation observations for arXiv:1810.02032."}