{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:QZOJLYIM6NL4TVZBTDBB2PCKY6","short_pith_number":"pith:QZOJLYIM","schema_version":"1.0","canonical_sha256":"865c95e10cf357c9d72198c21d3c4ac78dd1a2d220c4077cb4595b80d3121899","source":{"kind":"arxiv","id":"1908.03265","version":4},"attestation_state":"computed","paper":{"title":"On the Variance of the Adaptive Learning Rate and Beyond","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Haoming Jiang, Jianfeng Gao, Jiawei Han, Liyuan Liu, Pengcheng He, Weizhu Chen, XiaoDong Liu","submitted_at":"2019-08-08T20:51:17Z","abstract_excerpt":"The learning rate warmup heuristic achieves remarkable success in stabilizing training, accelerating convergence and improving generalization for adaptive stochastic optimization algorithms like RMSprop and Adam. Here, we study its mechanism in details. Pursuing the theory behind warmup, we identify a problem of the adaptive learning rate (i.e., it has problematically large variance in the early stage), suggest warmup works as a variance reduction technique, and provide both empirical and theoretical evidence to verify our hypothesis. We further propose RAdam, a new variant of Adam, by introdu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.03265","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-08T20:51:17Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"babc3347877d33cd22acc4e3dec52a87386dd46f42b024e023950848177c606d","abstract_canon_sha256":"86ea584269e8243e14959500d0fd6a28a03387172b3915c99bc9e93bdb218b99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:25:28.038927Z","signature_b64":"/TL1FU/1v7iiIez79oZEncLGkJSPN+N6q4qgTR0Lk29cnCwuMew/GNi/oyyhSJMsah1X6x01y7NVaXF+owtkAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"865c95e10cf357c9d72198c21d3c4ac78dd1a2d220c4077cb4595b80d3121899","last_reissued_at":"2026-07-05T03:25:28.038414Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:25:28.038414Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Variance of the Adaptive Learning Rate and Beyond","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Haoming Jiang, Jianfeng Gao, Jiawei Han, Liyuan Liu, Pengcheng He, Weizhu Chen, XiaoDong Liu","submitted_at":"2019-08-08T20:51:17Z","abstract_excerpt":"The learning rate warmup heuristic achieves remarkable success in stabilizing training, accelerating convergence and improving generalization for adaptive stochastic optimization algorithms like RMSprop and Adam. Here, we study its mechanism in details. Pursuing the theory behind warmup, we identify a problem of the adaptive learning rate (i.e., it has problematically large variance in the early stage), suggest warmup works as a variance reduction technique, and provide both empirical and theoretical evidence to verify our hypothesis. We further propose RAdam, a new variant of Adam, by introdu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.03265","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.03265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.03265","created_at":"2026-07-05T03:25:28.038485+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.03265v4","created_at":"2026-07-05T03:25:28.038485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.03265","created_at":"2026-07-05T03:25:28.038485+00:00"},{"alias_kind":"pith_short_12","alias_value":"QZOJLYIM6NL4","created_at":"2026-07-05T03:25:28.038485+00:00"},{"alias_kind":"pith_short_16","alias_value":"QZOJLYIM6NL4TVZB","created_at":"2026-07-05T03:25:28.038485+00:00"},{"alias_kind":"pith_short_8","alias_value":"QZOJLYIM","created_at":"2026-07-05T03:25:28.038485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14187","citing_title":"Zeta: Dual Whitening for Matrix Optimization via Coordinate-Adaptive Preconditioning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09658","citing_title":"Muon Learns More Robust and Transferable Features than Adam","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04662","citing_title":"Why Muon Outperforms Adam: A Curvature Perspective","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01032","citing_title":"Point spread function wavefront recovery from in-focus stellar observations","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21557","citing_title":"Scalable Reinforcement Learning via Adaptive Batch Scaling","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29255","citing_title":"Confidence-feedback-weighted graph matching network: online-offline laser-induced damage site matching under complex interference","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01665","citing_title":"TABX: A High-Throughput Sandbox Battle Simulator for Multi-Agent Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21557","citing_title":"Scalable Reinforcement Learning via Adaptive Batch Scaling","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15133","citing_title":"Building Deep Graph Predictors with Graph Imitation Learning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2310.14189","citing_title":"Improved Techniques for Training Consistency Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06558","citing_title":"Rapid training of Hamiltonian graph networks using random features","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2509.13389","citing_title":"From Next Token Prediction to (STRIPS) World Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.15816","citing_title":"On the Convergence of Muon and Beyond","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09178","citing_title":"Characterizing the Instrumental Profile of LAMOST","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2303.01469","citing_title":"Consistency Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12230","citing_title":"Neural Network-Based Virtual Wheel-Speed Sensor for Enhanced Low-Velocity State Estimation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2310.04378","citing_title":"Latent Consistency Models: Synthesizing High-Resolution Images with Few-Step Inference","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05115","citing_title":"Manifold Steering Reveals the Shared Geometry of Neural Network Representation and Behavior","ref_index":297,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02317","citing_title":"Anon: Extrapolating Adaptivity Beyond SGD and Adam","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00650","citing_title":"AdaMeZO: Adam-style Zeroth-Order Optimizer for LLM Fine-tuning Without Maintaining the Moments","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09809","citing_title":"Particle transformers for identifying Lorentz-boosted Higgs bosons decaying to a pair of W bosons","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01667","citing_title":"Deep neural networks with Fisher vector encoding for medical image classification","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08939","citing_title":"Delve into the Applicability of Advanced Optimizers for Multi-Task Learning","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6","json":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6.json","graph_json":"https://pith.science/api/pith-number/QZOJLYIM6NL4TVZBTDBB2PCKY6/graph.json","events_json":"https://pith.science/api/pith-number/QZOJLYIM6NL4TVZBTDBB2PCKY6/events.json","paper":"https://pith.science/paper/QZOJLYIM"},"agent_actions":{"view_html":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6","download_json":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6.json","view_paper":"https://pith.science/paper/QZOJLYIM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.03265&json=true","fetch_graph":"https://pith.science/api/pith-number/QZOJLYIM6NL4TVZBTDBB2PCKY6/graph.json","fetch_events":"https://pith.science/api/pith-number/QZOJLYIM6NL4TVZBTDBB2PCKY6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6/action/storage_attestation","attest_author":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6/action/author_attestation","sign_citation":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6/action/citation_signature","submit_replication":"https://pith.science/pith/QZOJLYIM6NL4TVZBTDBB2PCKY6/action/replication_record"}},"created_at":"2026-07-05T03:25:28.038485+00:00","updated_at":"2026-07-05T03:25:28.038485+00:00"}