{"as_of":"2026-08-11T02:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:469036ac41b77acf44afca34dc355364006800aa247e1af48ba52ad76bc50486","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":9,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":9,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T12:16:42.973908Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T14:49:54.902609Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2401.01335","last_updated":"2024-06-14T21:17:17Z","snapshot_observed_at":"2026-08-10T05:25:44.328250Z","submitted_at":"2024-01-02T18:53:13Z","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","version":3},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-14T23:00:20.720030Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2401.01335"},"observation_digest":"sha256:165ca02f293cf3c87375a1fbe83fc910fdc4f63f9841ff0077475dabe8b27fb5","observation_id":"e56068ef-b400-4a52-a3cc-95650784d892","resolution":{"observed_at":"2026-05-14T23:00:21.330637Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2601.22409","last_updated":"2026-05-12T23:48:38Z","snapshot_observed_at":"2026-08-02T15:20:04.793342Z","submitted_at":"2026-01-29T23:43:26Z","title":"Optimization, Generalization and Differential Privacy Bounds for Gradient Descent on Kolmogorov-Arnold Networks","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T09:34:00.409455Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2601.22409"},"observation_digest":"sha256:85f964c5c054d8e39b196bef2ae8be4dad7829863b48bc0bd5be94a16d311dc7","observation_id":"6dba53df-2264-4032-b059-820247e2d77d","resolution":{"observed_at":"2026-05-16T09:37:41.914956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2605.12648","last_updated":"2026-05-12T18:44:47Z","snapshot_observed_at":"2026-07-06T23:24:18.302402Z","submitted_at":"2026-05-12T18:44:47Z","title":"Population Risk Bounds for Kolmogorov-Arnold Networks Trained by DP-SGD with Correlated Noise","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-14T21:39:34.680193Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2605.12648"},"observation_digest":"sha256:0d61698c3e48ab8a70fce5fea7c6ed6e10e3b22f336b954928b25edde8147bc0","observation_id":"9246c17a-fb09-40af-9fd1-c520baa2e571","resolution":{"observed_at":"2026-05-14T21:43:00.870563Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2606.06764","last_updated":"2026-06-04T23:04:49Z","snapshot_observed_at":"2026-08-02T11:33:40.895435Z","submitted_at":"2026-06-04T23:04:49Z","title":"Optimal Rates for Generalization of Gradient Descent Methods with Deep Neural Networks","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T23:06:23.042883Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2606.06764"},"observation_digest":"sha256:85975864c01c2edd3c18c511a769aec6d8d7d4c7b63399c7ec90440c57453143","observation_id":"18b63de0-e0ae-4570-80c5-1575137f9f83","resolution":{"observed_at":"2026-07-02T15:57:07.396985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2606.06772","last_updated":"2026-07-29T13:42:12Z","snapshot_observed_at":"2026-08-07T13:57:17.299323Z","submitted_at":"2026-06-04T23:31:52Z","title":"Minimax-Optimal Generalization Bounds for Smooth Deep Neural Networks Trained by (Stochastic) Gradient Descent","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T23:03:52.889955Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2606.06772"},"observation_digest":"sha256:98b54d28bd993d7a525f56cd71e1be8f18d0d377a0bf57a9f39c4d00f431dca7","observation_id":"f76ee476-e345-4afd-9aa2-40c6a07aa2f6","resolution":{"observed_at":"2026-07-02T16:07:08.510992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-08-02T12:16:42.973908Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems.arXiv preprint arXiv:1905.09870, 2019","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2606.06772","last_updated":"2026-07-29T13:42:12Z","snapshot_observed_at":"2026-08-07T13:57:17.299323Z","submitted_at":"2026-06-04T23:31:52Z","title":"Minimax-Optimal Generalization Bounds for Smooth Deep Neural Networks Trained by (Stochastic) Gradient Descent","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T12:16:42.973908Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2606.06772"},"observation_digest":"sha256:073ca5a95965dd18a8b693f0677e4e03f7ba86e54f813d2f1cf6b63b55b30faf","observation_id":"f14d8b6b-c8eb-44e2-8ba6-cfc0d304c6f9","resolution":{"observed_at":"2026-08-02T12:16:42.973908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2606.10089","last_updated":"2026-06-08T19:16:32Z","snapshot_observed_at":"2026-08-02T05:36:24.589418Z","submitted_at":"2026-06-08T19:16:32Z","title":"A Theory on Flow Matching with Neural Networks","version":1},"reference_index":123,"source":"arxiv_source","source_observed_at":"2026-06-27T16:59:34.084575Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2606.10089"},"observation_digest":"sha256:39fce717102eadb35b3ae776aa387ea130ee64ee39d26d614d36dab826d7721b","observation_id":"4ed481b6-32df-442c-989b-7e1e45308d7d","resolution":{"observed_at":"2026-07-03T00:57:29.386764Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2606.26749","last_updated":"2026-06-25T08:33:34Z","snapshot_observed_at":"2026-08-01T16:00:20.311664Z","submitted_at":"2026-06-25T08:33:34Z","title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","version":1},"reference_index":272,"source":"arxiv_source","source_observed_at":"2026-06-26T05:14:07.208255Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2606.26749"},"observation_digest":"sha256:50800eb82ed28e14d3e6ca3b8f36bb8e35bf853e4b54ef1c01f45c98c43ca26e","observation_id":"15290d15-d8d8-4826-87ad-075ba7be7053","resolution":{"observed_at":"2026-07-04T13:29:51.148754Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","version":3},"cited_work":{"arxiv_id":"1905.09870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1905.09870","snapshot_observed_at":"2026-07-04T14:49:54.902609Z","title":"Gradient descent can learn less over-parameterized two-layer neural networks on classification problems","venue":null,"work_id":"3c14153d-16cf-45c5-92f8-fd04a748489e","year":1905},"citing_paper":{"arxiv_id":"2606.27142","last_updated":"2026-06-25T15:15:59Z","snapshot_observed_at":"2026-08-04T00:00:21.065840Z","submitted_at":"2026-06-25T15:15:59Z","title":"Estimation of High Dimensional Bounded Discrete Graphical Models via Regularized Generalized Score Matching","version":1},"reference_index":256,"source":"arxiv_source","source_observed_at":"2026-06-26T02:36:28.490582Z"},"links":{"cited_paper":"/paper/1905.09870","citing_paper":"/paper/2606.27142"},"observation_digest":"sha256:80612f9d8284bca70bf3cb8a9f2fb42ac6166fd0e0e52e36977787b1965a482a","observation_id":"74ac9a67-721b-41c0-9fea-69195c08c7f4","resolution":{"observed_at":"2026-07-04T14:49:54.906031Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/1905.09870/citation-record","integrity":"/paper/1905.09870/integrity","json":"/paper/1905.09870/citation-record.json","paper":"/paper/1905.09870"},"outbound":[],"paper":{"arxiv_id":"1905.09870","last_updated":"2020-03-18T13:59:10Z","latest_version":3,"primary_category":"stat.ML","snapshot_observed_at":"2026-07-06T07:55:05.948249Z","submitted_at":"2019-05-23T18:57:35Z","title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 9 inbound Pith citation observations for arXiv:1905.09870."}