{"as_of":"2026-08-13T09:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:37dea5bdc73d23d9291d132fa9370e42354b076fcb54a0c202fb1a9924968716","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":56,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T21:29:22.668886Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":608,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2303.01469","last_updated":"2023-05-31T06:17:10Z","snapshot_observed_at":"2026-08-13T00:53:42.623435Z","submitted_at":"2023-03-02T18:30:16Z","title":"Consistency Models","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-13T15:47:28.598205Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2303.01469"},"observation_digest":"sha256:afa85fb8abc7c208c9151f705e0b1a30dc532f5eb6a6fbd05120a8ef04cd560a","observation_id":"43527666-8bab-44bb-b011-16202f68355f","resolution":{"observed_at":"2026-05-13T15:47:28.659081Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2306.14048","last_updated":"2023-12-18T19:10:00Z","snapshot_observed_at":"2026-08-06T19:19:02.165567Z","submitted_at":"2023-06-24T20:11:14Z","title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","version":3},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-17T18:00:50.053377Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2306.14048"},"observation_digest":"sha256:1919878cfd14d5a115acc272922290c93dee9d62dacfc86400a0f3cc82bdc043","observation_id":"760fe633-d1de-41b2-9955-934cca7e5fef","resolution":{"observed_at":"2026-05-17T18:00:50.197627Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2310.04378","last_updated":"2023-10-06T17:11:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-06T17:11:58Z","title":"Latent Consistency Models: Synthesizing High-Resolution Images with Few-Step Inference","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-13T04:15:54.681913Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2310.04378"},"observation_digest":"sha256:553e9c812feeaacd76027713ba0068160f3e4e663cc8e76685b48a89a16f38d9","observation_id":"216fd231-8fdc-4802-a6af-9bb99ed6622e","resolution":{"observed_at":"2026-05-13T04:15:54.867502Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2310.14189","last_updated":"2023-10-22T05:33:38Z","snapshot_observed_at":"2026-08-03T16:40:35.231838Z","submitted_at":"2023-10-22T05:33:38Z","title":"Improved Techniques for Training Consistency Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T05:04:10.600062Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2310.14189"},"observation_digest":"sha256:2cc4bd0f7c894e489ce3374a33f9843703ea487323e2c37d783feb7f019a9cd4","observation_id":"18eb65d8-34fc-45ea-a560-afdec281900c","resolution":{"observed_at":"2026-05-21T05:04:10.662000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-12T14:11:16.550216Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2411.15638","last_updated":"2025-03-26T22:36:18Z","snapshot_observed_at":"2026-08-12T14:02:43.626556Z","submitted_at":"2024-11-23T19:30:56Z","title":"Learning state and proposal dynamics in state-space models using differentiable particle filters and neural networks","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T14:11:16.550216Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2411.15638"},"observation_digest":"sha256:ce44cb352f683ecdb470e32c4a949624bda697fa67c237dae8e6839152b78f7b","observation_id":"ab8f389a-b751-480d-901d-8f36cad52591","resolution":{"observed_at":"2026-08-12T14:11:16.550216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-12T14:02:39.847015Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.15795","last_updated":"2024-12-16T11:06:00Z","snapshot_observed_at":"2026-08-12T13:51:16.889059Z","submitted_at":"2024-11-24T11:46:47Z","title":"Beyond adaptive gradient: Fast-Controlled Minibatch Algorithm for large-scale optimization","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T14:02:39.847015Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2411.15795"},"observation_digest":"sha256:a4bad4c07e5c176b4fdf7a6184d633c0d3eda5c6804bae6f842dc680c7272589","observation_id":"e000a8d5-f4b0-4d2b-88e4-efff3ef24cd6","resolution":{"observed_at":"2026-08-12T14:02:39.847015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-12T21:29:22.668886Z","title":"Liu , author H","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2411.17709","last_updated":"2024-11-13T16:15:48Z","snapshot_observed_at":"2026-08-12T21:23:00.106800Z","submitted_at":"2024-11-13T16:15:48Z","title":"Quantity versus Diversity: Influence of Data on Detecting EEG Pathology with Advanced ML Models","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-12T21:29:22.668886Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2411.17709"},"observation_digest":"sha256:fa292cac21e3d57d2007ee603039ad0b8e0e6bea604a7c0a49d294ed2dd488b6","observation_id":"4a93ff98-854e-4555-9890-62befd0919e2","resolution":{"observed_at":"2026-08-12T21:29:22.668886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-12T06:03:51.315324Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2411.19647","last_updated":"2025-06-04T14:58:21Z","snapshot_observed_at":"2026-08-12T05:56:53.862811Z","submitted_at":"2024-11-29T12:00:27Z","title":"CAdam: Confidence-Based Optimization for Online Learning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T06:03:51.315324Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2411.19647"},"observation_digest":"sha256:edf461dceccc84be22eb87822045998de2095f65f52cd8e38be7b6f73d81caba","observation_id":"7e4ce5f8-ae17-43a7-a779-efda4c0986a0","resolution":{"observed_at":"2026-08-12T06:03:51.315324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-12T05:26:44.820508Z","title":", Jiang, H","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.00486","last_updated":"2024-11-30T13:58:41Z","snapshot_observed_at":"2026-08-12T05:17:48.436662Z","submitted_at":"2024-11-30T13:58:41Z","title":"Automatic Differentiation-based Full Waveform Inversion with Flexible Workflows","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-12T05:26:44.820508Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2412.00486"},"observation_digest":"sha256:ddf11c23c436e241bd660e971e521cbd5b63039cb846a87b247256fe2ef4fd1e","observation_id":"f72ffc1a-e400-4bde-a3a1-d8ddbcfa11fe","resolution":{"observed_at":"2026-08-12T05:26:44.820508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-11T19:24:03.709142Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.06686","last_updated":"2024-12-09T17:28:29Z","snapshot_observed_at":"2026-08-12T04:19:48.092728Z","submitted_at":"2024-12-09T17:28:29Z","title":"Some Best Practices in Operator Learning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T19:24:03.709142Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2412.06686"},"observation_digest":"sha256:291cc7d2e8d59314e2dfdcd2e96d73cc0649faa60ecde347dc916fd32e311996","observation_id":"355fd820-b7cc-4138-a443-29807ff5ffbf","resolution":{"observed_at":"2026-08-11T19:24:03.709142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-11T14:41:00.366975Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2412.11768","last_updated":"2024-12-17T09:30:44Z","snapshot_observed_at":"2026-08-11T14:58:47.107932Z","submitted_at":"2024-12-16T13:41:37Z","title":"No More Adam: Learning Rate Scaling at Initialization is All You Need","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T14:41:00.366975Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2412.11768"},"observation_digest":"sha256:ef90f711198b2950566b80916d798e5f21f45fc3cc8edea661f8f9feb622f380","observation_id":"ddff11e7-46f3-4f80-83da-3fa88889e118","resolution":{"observed_at":"2026-08-11T14:41:00.366975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-11T11:32:54.039495Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2412.16243","last_updated":"2024-12-19T20:52:10Z","snapshot_observed_at":"2026-08-12T23:14:03.601624Z","submitted_at":"2024-12-19T20:52:10Z","title":"Bag of Tricks for Multimodal AutoML with Image, Text, and Tabular Data","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T11:32:54.039495Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2412.16243"},"observation_digest":"sha256:72e22d0164d5faa24c3999d13c31e6d3034c1365cf98942d5995a11344886c5b","observation_id":"7aa1e192-1da6-423c-b252-6cab0852ad95","resolution":{"observed_at":"2026-08-11T11:32:54.039495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-11T10:24:04.385407Z","title":"On the vari- ance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265, 2019","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2412.16715","last_updated":"2024-12-21T17:57:12Z","snapshot_observed_at":"2026-08-11T14:58:49.399700Z","submitted_at":"2024-12-21T17:57:12Z","title":"From Histopathology Images to Cell Clouds: Learning Slide Representations with Hierarchical Cell Transformer","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T10:24:04.385407Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2412.16715"},"observation_digest":"sha256:46b87652e34272787a86000b110437cea0c780d632f0c290b90f8f5f836eb950","observation_id":"128c70fd-0c9b-4d1b-8255-ca8d832252b5","resolution":{"observed_at":"2026-08-11T10:24:04.385407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-10T23:14:17.622839Z","title":"On the variance of the adaptive learning rate and beyond,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2412.20830","last_updated":"2024-12-30T09:53:26Z","snapshot_observed_at":"2026-08-11T14:57:23.481856Z","submitted_at":"2024-12-30T09:53:26Z","title":"ReFlow6D: Refraction-Guided Transparent Object 6D Pose Estimation via Intermediate Representation Learning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T23:14:17.622839Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2412.20830"},"observation_digest":"sha256:81b900f333bdb352c96853ed5e500193c93ae81a8bfbabbc7bcfbfb34bd5c2ef","observation_id":"9d0f8ad0-3632-4a27-9641-b8111129220b","resolution":{"observed_at":"2026-08-10T23:14:17.622839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-10T17:01:13.908163Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2501.12670","last_updated":"2025-06-19T15:01:04Z","snapshot_observed_at":"2026-08-13T06:56:55.902105Z","submitted_at":"2025-01-22T06:10:27Z","title":"Celo: Training Versatile Learned Optimizers on a Compute Diet","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-10T17:01:13.908163Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2501.12670"},"observation_digest":"sha256:40002abc5ad8caee5cb3104b35fe5e5c1fbf64821b90d65b56f434d576956329","observation_id":"101104e3-08fe-4b5d-b2a8-05102d73849c","resolution":{"observed_at":"2026-08-10T17:01:13.908163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-10T16:51:20.102451Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.12815","last_updated":"2025-01-22T11:46:28Z","snapshot_observed_at":"2026-08-11T04:28:31.652084Z","submitted_at":"2025-01-22T11:46:28Z","title":"Certified Guidance for Planning with Deep Generative Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T16:51:20.102451Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2501.12815"},"observation_digest":"sha256:e00c5fc7a5bccde9137411ad74a417122b323067af7830fa44c57fd2eeefbab1","observation_id":"41337159-851c-4c38-8ebb-f4896e234754","resolution":{"observed_at":"2026-08-10T16:51:20.102451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-10T15:46:20.139219Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.13683","last_updated":"2025-01-23T14:10:02Z","snapshot_observed_at":"2026-08-11T14:57:18.137763Z","submitted_at":"2025-01-23T14:10:02Z","title":"Unlearning Clients, Features and Samples in Vertical Federated Learning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T15:46:20.139219Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2501.13683"},"observation_digest":"sha256:bd84dcfafec14e7b8f1df2bd8be98c8c66f008bc6566713e6aba43521df843c6","observation_id":"6c36cb10-3e04-4742-978e-6ea2c9d7138b","resolution":{"observed_at":"2026-08-10T15:46:20.139219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-09T20:27:53.458185Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.19364","last_updated":"2025-07-26T09:31:22Z","snapshot_observed_at":"2026-08-11T15:52:27.132223Z","submitted_at":"2025-01-31T18:14:28Z","title":"CoSTI: Consistency Models for (a faster) Spatio-Temporal Imputation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-09T20:27:53.458185Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2501.19364"},"observation_digest":"sha256:9d5c5e42d594632372e6b4048d7589ead709706ad46fd30f5387501d7ddd8e11","observation_id":"f125ef0f-c08d-4f99-a0e3-9afff26d105e","resolution":{"observed_at":"2026-08-09T20:27:53.458185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2506.06558","last_updated":"2026-03-02T10:41:47Z","snapshot_observed_at":"2026-08-11T03:08:47.597682Z","submitted_at":"2025-06-06T22:10:05Z","title":"Rapid training of Hamiltonian graph networks using random features","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-19T10:17:00.343410Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2506.06558"},"observation_digest":"sha256:08921722ee9b54fed190494c378a24d61b5d7a30a4f9e90ce6762d0da5bd02e5","observation_id":"9b8548ed-6392-44bd-aba0-c0307459498a","resolution":{"observed_at":"2026-05-19T10:17:15.702752Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-06T20:07:26.301836Z","title":"2019, arXiv e-prints, arXiv:1908.03265, 10.48550/arXiv.1908.03265","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.03742","last_updated":"2025-08-04T14:23:51Z","snapshot_observed_at":"2026-08-11T04:18:49.742949Z","submitted_at":"2025-07-04T18:00:00Z","title":"Deep Potential: Recovering the gravitational potential and local pattern speed in the solar neighborhood with GDR3 using normalizing flows","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-06T20:07:26.301836Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2507.03742"},"observation_digest":"sha256:863be4ab8bb0393d4e3b398d63b5eb48fed2fdb4574a7afc42f7e226c1aade81","observation_id":"05fd9f67-538a-4a5b-975b-c91c9b65fd08","resolution":{"observed_at":"2026-08-06T20:07:26.301836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-06T19:10:16.591929Z","title":"On the variance of the adaptive learning rate and beyond,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2507.06464","last_updated":"2025-07-09T00:47:37Z","snapshot_observed_at":"2026-08-13T07:37:39.690725Z","submitted_at":"2025-07-09T00:47:37Z","title":"SoftSignSGD(S3): An Enhanced Optimizer for Practical DNN Training and Loss Spikes Minimization Beyond Adam","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:10:16.591929Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2507.06464"},"observation_digest":"sha256:7d8b632fd917c5b00d0957f36d71f40df3d8cb2634e896a575f778fb8e3ee235","observation_id":"218ef676-cdc5-4fff-9afb-a88f3cad0e79","resolution":{"observed_at":"2026-08-06T19:10:16.591929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-06T16:43:51.780376Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2507.12828","last_updated":"2025-07-23T02:14:58Z","snapshot_observed_at":"2026-08-13T06:52:28.432966Z","submitted_at":"2025-07-17T06:37:45Z","title":"Feature-Enhanced TResNet for Fine-Grained Food Image Classification","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T16:43:51.780376Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2507.12828"},"observation_digest":"sha256:d1b2e6a289b5769e503297f07ec9f68c3f7b03862a84f73e4f636d1d48189c1b","observation_id":"9e6a5159-a688-48c3-9649-7e32726407b1","resolution":{"observed_at":"2026-08-06T16:43:51.780376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-06T06:01:11.597771Z","title":"On the variance of the adaptive learning rate and beyond,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.01001","last_updated":"2025-08-01T18:10:49Z","snapshot_observed_at":"2026-08-09T20:22:50.977080Z","submitted_at":"2025-08-01T18:10:49Z","title":"Criticality analysis of nuclear binding energy neural networks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T06:01:11.597771Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2508.01001"},"observation_digest":"sha256:6be95ee9913ec1cfb0dbf527dae214188c830310e543ff158c22a5b54068da46","observation_id":"80f58d11-afa8-4b7e-b467-8b8a79d71466","resolution":{"observed_at":"2026-08-06T06:01:11.597771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2509.13389","last_updated":"2026-05-25T16:41:30Z","snapshot_observed_at":"2026-08-12T15:01:25.404374Z","submitted_at":"2025-09-16T14:03:58Z","title":"From Next Token Prediction to (STRIPS) World Models","version":6},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-18T16:21:22.805473Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2509.13389"},"observation_digest":"sha256:ac2215967d8326bfdf71140c82f79caf5b91d8cc59b919670674f4f16d2b8e66","observation_id":"d42c4cf1-b7c5-41ff-983f-823e41ef7f6f","resolution":{"observed_at":"2026-05-18T16:21:36.462305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-04T16:41:18.775357Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.13389","last_updated":"2026-05-25T16:41:30Z","snapshot_observed_at":"2026-08-12T15:01:25.404374Z","submitted_at":"2025-09-16T14:03:58Z","title":"From Next Token Prediction to (STRIPS) World Models","version":7},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-04T16:41:18.775357Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2509.13389"},"observation_digest":"sha256:5d874177edd241777dd46efcb53148cbbe956f33f8c210928f244e882ac49ffe","observation_id":"37157420-cf3a-4228-9c3f-f79600f54f5b","resolution":{"observed_at":"2026-08-04T16:41:18.775357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2509.15816","last_updated":"2026-05-10T06:58:46Z","snapshot_observed_at":"2026-08-12T12:31:17.777645Z","submitted_at":"2025-09-19T09:43:37Z","title":"On the Convergence of Muon and Beyond","version":5},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-18T15:56:30.602824Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2509.15816"},"observation_digest":"sha256:fca7a6ac86be8942a9f4ee52cca4426d491936d3dd73622fb70c42514056ef6c","observation_id":"53416c7b-10c1-493a-a408-472c2dd13e13","resolution":{"observed_at":"2026-05-18T15:56:34.102980Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-04T12:38:58.819676Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2510.03164","last_updated":"2026-06-28T16:14:54Z","snapshot_observed_at":"2026-08-04T12:38:48.240318Z","submitted_at":"2025-10-03T16:35:56Z","title":"Why Do We Need Warm-up? A Theoretical Perspective","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-04T12:38:58.819676Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2510.03164"},"observation_digest":"sha256:9cc5bafb449e2c0e2be71f52e1d222c16d8613eb2d0c0ff7ec3c1f90f27fb429","observation_id":"4a969c6c-cba2-4130-8c93-7308ff18d135","resolution":{"observed_at":"2026-08-04T12:38:58.819676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2601.15133","last_updated":"2026-05-20T17:50:47Z","snapshot_observed_at":"2026-07-06T22:42:36.961014Z","submitted_at":"2026-01-21T16:07:17Z","title":"Building Deep Graph Predictors with Graph Imitation Learning","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-21T15:44:58.327091Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2601.15133"},"observation_digest":"sha256:ba9c6ff6b6b4021e9172eb48cfabe2291f6ed7ed247af2d5b2cd514e516618e6","observation_id":"8606aaef-40c8-497f-ba60-d25e8cb2dc9c","resolution":{"observed_at":"2026-05-21T15:45:18.726240Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2602.01665","last_updated":"2026-05-27T04:20:45Z","snapshot_observed_at":"2026-08-03T05:37:34.993758Z","submitted_at":"2026-02-02T05:34:38Z","title":"TABX: A High-Throughput Sandbox Battle Simulator for Multi-Agent Reinforcement Learning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-25T07:41:48.634351Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2602.01665"},"observation_digest":"sha256:971fce56f97b3c28ba5498b94f3036797801f8d2e673b4ba08c8729e406eddd7","observation_id":"1cceda00-a532-4936-a3b8-d125000f1e9a","resolution":{"observed_at":"2026-05-25T07:45:29.768850Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-03T05:37:38.449157Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2602.01665","last_updated":"2026-05-27T04:20:45Z","snapshot_observed_at":"2026-08-03T05:37:34.993758Z","submitted_at":"2026-02-02T05:34:38Z","title":"TABX: A High-Throughput Sandbox Battle Simulator for Multi-Agent Reinforcement Learning","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T05:37:38.449157Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2602.01665"},"observation_digest":"sha256:ec88244f8048b5d9a92ea4f2457778e00cd90af5277ec4da66088ebab3e2021c","observation_id":"6999988d-3e60-41a8-b3a6-590f5e3cf82e","resolution":{"observed_at":"2026-08-03T05:37:38.449157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2603.09178","last_updated":"2026-04-13T02:12:38Z","snapshot_observed_at":"2026-08-13T01:43:30.225341Z","submitted_at":"2026-03-10T04:32:43Z","title":"Characterizing the Instrumental Profile of LAMOST","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T14:09:12.016309Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2603.09178"},"observation_digest":"sha256:ea47a18899ba9a2abb4649849866382f1711f4a5c251989c7d156c62d8017d9a","observation_id":"becc2260-cb50-4253-b2ff-542c3fcc3f4b","resolution":{"observed_at":"2026-05-15T14:10:03.181485Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2604.06789","last_updated":"2026-04-08T07:57:05Z","snapshot_observed_at":"2026-07-06T22:55:11.560398Z","submitted_at":"2026-04-08T07:57:05Z","title":"Video-guided Machine Translation with Global Video Context","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T18:02:31.374359Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2604.06789"},"observation_digest":"sha256:6b3e7b89362a9810a2d32366b1f6dea9a822d975ef15c7dc13cf222a4c95a55f","observation_id":"6b7d67b1-d767-4ff2-851f-29266263f10f","resolution":{"observed_at":"2026-05-11T05:36:01.141997Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2604.08939","last_updated":"2026-04-10T04:15:31Z","snapshot_observed_at":"2026-08-11T08:54:52.283153Z","submitted_at":"2026-04-10T04:15:31Z","title":"Delve into the Applicability of Advanced Optimizers for Multi-Task Learning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T16:42:48.902337Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2604.08939"},"observation_digest":"sha256:948f6212af1533b95b57fc3c26238546a40c0efbcf64b6d65926aaedf22c1a06","observation_id":"e23ddd4e-e4dc-4372-856d-6bf8144e9d3e","resolution":{"observed_at":"2026-05-11T08:20:59.066359Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2604.09809","last_updated":"2026-04-10T18:33:27Z","snapshot_observed_at":"2026-08-02T06:33:22.153300Z","submitted_at":"2026-04-10T18:33:27Z","title":"Particle transformers for identifying Lorentz-boosted Higgs bosons decaying to a pair of W bosons","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-10T16:18:12.735038Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2604.09809"},"observation_digest":"sha256:a834c5a604e3b0b274fe39c5e24262745314731baa19c7f2c19b045cadac1dfa","observation_id":"c01c0deb-37e6-4e97-bab7-0c246942c6dc","resolution":{"observed_at":"2026-05-11T09:01:00.835334Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2604.22838","last_updated":"2026-04-21T06:27:18Z","snapshot_observed_at":"2026-08-13T01:18:51.805877Z","submitted_at":"2026-04-21T06:27:18Z","title":"Neural Network Optimization Reimagined: Decoupled Techniques for Scratch and Fine-Tuning","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-10T03:26:09.751493Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2604.22838"},"observation_digest":"sha256:267f422931321c131f00e61005263ac0214df521d48d19b4362703f23ea1e1a1","observation_id":"8ae7bd9e-455d-4181-9297-823f6433ac9f","resolution":{"observed_at":"2026-05-10T03:29:22.069640Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2605.00650","last_updated":"2026-05-01T13:31:35Z","snapshot_observed_at":"2026-08-12T12:26:48.658869Z","submitted_at":"2026-05-01T13:31:35Z","title":"AdaMeZO: Adam-style Zeroth-Order Optimizer for LLM Fine-tuning Without Maintaining the Moments","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-09T19:50:50.653184Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2605.00650"},"observation_digest":"sha256:7cd63f614b7ae0533c66a776748bbc5fc3a7410fbe2b655b4f0125c66f12ca99","observation_id":"59fe7434-96b1-48a5-8a05-df485960b583","resolution":{"observed_at":"2026-05-11T15:31:07.615287Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2605.01667","last_updated":"2026-05-03T01:38:37Z","snapshot_observed_at":"2026-08-11T04:48:35.731312Z","submitted_at":"2026-05-03T01:38:37Z","title":"Deep neural networks with Fisher vector encoding for medical image classification","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T16:22:00.015928Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2605.01667"},"observation_digest":"sha256:17c5f12d217f1e1a391b5b75a46249c915a6fab803372b0ac7cece9a45176ffa","observation_id":"e984c67a-2fbf-4c5a-a644-babd64eadc37","resolution":{"observed_at":"2026-05-11T09:00:58.899036Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2605.02317","last_updated":"2026-05-06T13:42:00Z","snapshot_observed_at":"2026-08-13T03:12:28.636236Z","submitted_at":"2026-05-04T08:14:51Z","title":"Anon: Extrapolating Adaptivity Beyond SGD and Adam","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-09T16:25:36.073417Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2605.02317"},"observation_digest":"sha256:83f5ae1d299c606d7e30b6e9ad41415a744ba0744c329ea9dd212257ae265303","observation_id":"4e6ec95d-267c-4f21-8d9a-e2e9a6cd9a9c","resolution":{"observed_at":"2026-05-11T16:31:09.614347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2605.05115","last_updated":"2026-05-06T16:46:03Z","snapshot_observed_at":"2026-08-07T10:12:24.411121Z","submitted_at":"2026-05-06T16:46:03Z","title":"Manifold Steering Reveals the Shared Geometry of Neural Network Representation and Behavior","version":1},"reference_index":297,"source":"arxiv_source","source_observed_at":"2026-05-08T17:47:09.591001Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2605.05115"},"observation_digest":"sha256:554eca4d8e1e9881bc3609d20dcd834e2949a4dcfcbdf4cae200733d09780db5","observation_id":"00736419-3d20-4b7a-a440-5d677ac61205","resolution":{"observed_at":"2026-05-11T17:16:06.693844Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2605.12230","last_updated":"2026-05-12T15:05:32Z","snapshot_observed_at":"2026-08-11T15:46:37.563863Z","submitted_at":"2026-05-12T15:05:32Z","title":"Neural Network-Based Virtual Wheel-Speed Sensor for Enhanced Low-Velocity State Estimation","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-13T04:32:39.053186Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2605.12230"},"observation_digest":"sha256:24a7640fbe49575279a09b698743bfd2a341059139f0e17420c0cdd32392166c","observation_id":"2918b059-9733-48f5-a3cd-47bd6101ba2e","resolution":{"observed_at":"2026-05-13T04:37:15.837211Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2605.21557","last_updated":"2026-06-04T12:29:12Z","snapshot_observed_at":"2026-07-06T23:32:01.202542Z","submitted_at":"2026-05-20T13:46:22Z","title":"Scalable Reinforcement Learning via Adaptive Batch Scaling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T00:21:15.933174Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2605.21557"},"observation_digest":"sha256:8391757288e18526a51c60f01dbabb97b1e6a43ac4ab7789e1041157fd08131e","observation_id":"59e925a4-f420-47fe-a40e-8169504af7f9","resolution":{"observed_at":"2026-05-22T00:24:27.799007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2605.21557","last_updated":"2026-06-04T12:29:12Z","snapshot_observed_at":"2026-07-06T23:32:01.202542Z","submitted_at":"2026-05-20T13:46:22Z","title":"Scalable Reinforcement Learning via Adaptive Batch Scaling","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T17:17:59.698127Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2605.21557"},"observation_digest":"sha256:53952c026e4ccf49c50c1b576d121336302228847a8250bf79392bc21ca597ec","observation_id":"200df501-70ac-4759-a340-d81bee41b7fe","resolution":{"observed_at":"2026-06-30T17:24:57.664358Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2606.04662","last_updated":"2026-06-03T09:40:30Z","snapshot_observed_at":"2026-08-12T19:27:04.962406Z","submitted_at":"2026-06-03T09:40:30Z","title":"Why Muon Outperforms Adam: A Curvature Perspective","version":1},"reference_index":169,"source":"arxiv_source","source_observed_at":"2026-06-28T07:04:21.012269Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2606.04662"},"observation_digest":"sha256:a09032a5e93bad01ec744394e23f104bb6e13413db11b6bc811427565a24a104","observation_id":"431879f9-244d-4bb3-bfd4-38146499c5ea","resolution":{"observed_at":"2026-07-02T07:06:44.922851Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2606.09658","last_updated":"2026-06-08T15:42:54Z","snapshot_observed_at":"2026-08-12T20:25:49.594664Z","submitted_at":"2026-06-08T15:42:54Z","title":"Muon Learns More Robust and Transferable Features than Adam","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-06-27T17:08:30.717799Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2606.09658"},"observation_digest":"sha256:9249183a177184411a3a2048bd583018d9ed27cac2f7b6e23a923c5fc92dcdd0","observation_id":"a73ac89d-0f45-4267-a7d4-3024d2c63041","resolution":{"observed_at":"2026-07-03T00:27:30.094776Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2606.14187","last_updated":"2026-06-16T11:02:08Z","snapshot_observed_at":"2026-08-08T02:01:14.681548Z","submitted_at":"2026-06-12T07:10:17Z","title":"Zeta: Dual Whitening for Matrix Optimization via Coordinate-Adaptive Preconditioning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T05:03:06.472027Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2606.14187"},"observation_digest":"sha256:1caccc59b2dca7021067ffb13427b0d9460f7e3255aff03679b9f155066eee74","observation_id":"5b67d29d-83ba-4c55-af17-ac27193e4821","resolution":{"observed_at":"2026-07-03T16:38:41.024926Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2606.25971","last_updated":"2026-07-17T13:42:08Z","snapshot_observed_at":"2026-08-02T10:13:56.529128Z","submitted_at":"2026-06-24T15:40:26Z","title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","version":1},"reference_index":166,"source":"arxiv_source","source_observed_at":"2026-06-25T20:05:09.179627Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2606.25971"},"observation_digest":"sha256:bffb1eb48713198131da6073db7f0bd17f4ecceb1272a073ca1d789dcc1c4dfc","observation_id":"8e1790d6-d0bb-4204-a2d9-f13af6cbcd0c","resolution":{"observed_at":"2026-07-04T20:30:07.698743Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-02T10:14:08.871888Z","title":"On the variance of the adaptive learning rate and beyond","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.25971","last_updated":"2026-07-17T13:42:08Z","snapshot_observed_at":"2026-08-02T10:13:56.529128Z","submitted_at":"2026-06-24T15:40:26Z","title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-02T10:14:08.871888Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2606.25971"},"observation_digest":"sha256:b207f1ba35aaa1c8e08907c50809b1ee19d5e92140d1a32c79e6d87c8554fdd0","observation_id":"bf816f9e-848a-4f3d-a30e-92ee06128d26","resolution":{"observed_at":"2026-08-02T10:14:08.871888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2606.29255","last_updated":"2026-06-28T07:53:16Z","snapshot_observed_at":"2026-08-02T15:16:00.907137Z","submitted_at":"2026-06-28T07:53:16Z","title":"Confidence-feedback-weighted graph matching network: online-offline laser-induced damage site matching under complex interference","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-30T08:09:50.192952Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2606.29255"},"observation_digest":"sha256:6453a6e5e0c2687ac02daca7131c71afc14b5c64f3d792a95a0c48ef774c5ed5","observation_id":"8aa828d9-8e2a-41c4-81a9-a577794beac8","resolution":{"observed_at":"2026-06-30T08:14:25.179388Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":"1908.03265","doi":"10.48550/arxiv.1908.03265","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265","venue":"arXiv (Cornell University)","work_id":"7acbf612-b1e2-43af-bfe5-3a9d75e45cb6","year":1908},"citing_paper":{"arxiv_id":"2607.01032","last_updated":"2026-07-01T15:00:21Z","snapshot_observed_at":"2026-08-02T08:29:05.462200Z","submitted_at":"2026-07-01T15:00:21Z","title":"Point spread function wavefront recovery from in-focus stellar observations","version":1},"reference_index":187,"source":"arxiv_source","source_observed_at":"2026-07-02T05:34:06.325999Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.01032"},"observation_digest":"sha256:9a22053eb7fa2a18b18b544fd87d626c0d49103bf3f84df289c5169c4ba445ae","observation_id":"3909f6fb-bb73-4b1a-8d41-6a0245ba64c8","resolution":{"observed_at":"2026-07-02T05:36:40.238509Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T18:08:20.655659+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-07-11T22:10:49.683444Z","title":"On the variance of the adaptive learning rate and beyond.arXiv preprint arXiv:1908.03265, 2019","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.04033","last_updated":"2026-07-04T21:27:05Z","snapshot_observed_at":"2026-08-07T05:07:10.107694Z","submitted_at":"2026-07-04T21:27:05Z","title":"OmniOpt: Taxonomy, Geometry, and Benchmarking of Modern Optimizers","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-07-11T22:10:49.683444Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.04033"},"observation_digest":"sha256:489cb4bb054531ea47658205b8a44352ca65ba99577e803297d13c9c5e315a45","observation_id":"9acfc265-1011-43dd-a2a7-b148e9610e63","resolution":{"observed_at":"2026-07-11T22:10:49.683444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-07-11T20:46:05.467029Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.04233","last_updated":"2026-07-05T11:11:40Z","snapshot_observed_at":"2026-08-06T18:57:36.945231Z","submitted_at":"2026-07-05T11:11:40Z","title":"Unified convergence analysis for gradient descent optimization methods in the training of deep neural networks","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-07-11T20:46:05.467029Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.04233"},"observation_digest":"sha256:daa1ea8f957523aeaffd068491aef0fdbd350beb246acfa775e369346529d268","observation_id":"e6b32ccb-1afa-420f-801f-d94cfd168b29","resolution":{"observed_at":"2026-07-11T20:46:05.467029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-01T16:19:27.144952Z","title":null,"venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.18058","last_updated":"2026-07-20T15:26:50Z","snapshot_observed_at":"2026-08-12T16:49:56.120466Z","submitted_at":"2026-07-20T15:26:50Z","title":"A machine-learned probability distribution in the phase space of turbulent channel flow for synthetic turbulence and flow reconstruction","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-01T16:19:27.144952Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.18058"},"observation_digest":"sha256:170e7075dcd52b4ee145e6536422f0ed77dda713e10092e8ee9102d32047a849","observation_id":"f9e1f0fb-c62e-4366-b4a9-d290f097a92e","resolution":{"observed_at":"2026-08-01T16:19:27.144952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-01T08:51:11.204445Z","title":"arXiv preprint arXiv:1908.03265 , year=","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.21009","last_updated":"2026-07-23T07:47:31Z","snapshot_observed_at":"2026-08-10T14:50:28.814846Z","submitted_at":"2026-07-23T07:47:31Z","title":"GCR Spectra Reconstructed with Neutron Monitor Yield Function and Artificial Neural Networks: Comparison of Two Methods","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-01T08:51:11.204445Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.21009"},"observation_digest":"sha256:245bd4e59b18dfcb79eed0c839d60dcb71e1b0280548e2aa2a3ae47c4051a635","observation_id":"a8553a71-a379-47ab-a86f-bd17140d0fb5","resolution":{"observed_at":"2026-08-01T08:51:11.204445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-01T04:40:18.200294Z","title":null,"venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.22499","last_updated":"2026-07-24T17:11:47Z","snapshot_observed_at":"2026-08-08T05:09:53.327026Z","submitted_at":"2026-07-24T17:11:47Z","title":"Payne4GAIN: NLTE Corrections for Red Giants in Milky Way Mapper using H-Band Neural Network Emulators","version":1},"reference_index":152,"source":"arxiv_source","source_observed_at":"2026-08-01T04:40:18.200294Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.22499"},"observation_digest":"sha256:6dbbdc4baaee3219f2a3f991fd49efcf86ef9bd26115329303c4ca7ebc6c6696","observation_id":"4aa982dc-4ad3-4a8a-aace-348d101318c1","resolution":{"observed_at":"2026-08-01T04:40:18.200294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-07-30T18:58:28.223966Z","title":"arXiv preprint arXiv:1908.03265 , year=","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.26860","last_updated":"2026-07-29T12:44:19Z","snapshot_observed_at":"2026-08-12T23:39:20.547745Z","submitted_at":"2026-07-29T12:44:19Z","title":"Amortized Moment Matching for Visual Generation","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-07-30T18:58:28.223966Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.26860"},"observation_digest":"sha256:780921fbb69e97a4e936b91f8d2988b9dfe2e0476dc6ecbf7253bfa57a4aab5f","observation_id":"320d20e7-082b-4fe6-a440-833a7c068a46","resolution":{"observed_at":"2026-07-30T18:58:28.223966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.03265","snapshot_observed_at":"2026-08-01T04:27:10.763773Z","title":"On the variance of the adaptive learning rate and beyond,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.27635","last_updated":"2026-07-30T03:45:38Z","snapshot_observed_at":"2026-08-12T19:51:09.098626Z","submitted_at":"2026-07-30T03:45:38Z","title":"HealthCAT: An Interpretable Encoder-only Transformer Framework for Health Indicator Prediction and Temporal Interpretation of Wearable Sensor Data","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-01T04:27:10.763773Z"},"links":{"cited_paper":"/paper/1908.03265","citing_paper":"/paper/2607.27635"},"observation_digest":"sha256:e8689c6065ad415cab44b5172eab82ad1653720a1c8e9cf083d331e12429013a","observation_id":"30ab29ce-e1d7-4d81-a48a-c360c4b99e42","resolution":{"observed_at":"2026-08-01T04:27:10.763773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1908.03265/citation-record","integrity":"/paper/1908.03265/integrity","json":"/paper/1908.03265/citation-record.json","paper":"/paper/1908.03265"},"outbound":[],"paper":{"arxiv_id":"1908.03265","last_updated":"2021-10-26T02:48:30Z","latest_version":4,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-11T14:56:04.364712Z","submitted_at":"2019-08-08T20:51:17Z","title":"On the Variance of the Adaptive Learning Rate and Beyond"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 56 inbound Pith citation observations for arXiv:1908.03265."}