{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:2B6CBYER7NUHGLCAS4SGQRUTUT","short_pith_number":"pith:2B6CBYER","schema_version":"1.0","canonical_sha256":"d07c20e091fb68732c409724684693a4c8c4fe7dca4aec4b070464b77d3b462e","source":{"kind":"arxiv","id":"2109.08282","version":1},"attestation_state":"computed","paper":{"title":"AdaLoss: A computationally-efficient and provably convergent adaptive gradient method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Rachel Ward, Simon Du, Xiaoxia Wu, Yuege Xie","submitted_at":"2021-09-17T01:45:25Z","abstract_excerpt":"We propose a computationally-friendly adaptive learning rate schedule, \"AdaLoss\", which directly uses the information of the loss function to adjust the stepsize in gradient descent methods. We prove that this schedule enjoys linear convergence in linear regression. Moreover, we provide a linear convergence guarantee over the non-convex regime, in the context of two-layer over-parameterized neural networks. If the width of the first-hidden layer in the two-layer networks is sufficiently large (polynomially), then AdaLoss converges robustly \\emph{to the global minimum} in polynomial time. We nu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.08282","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2021-09-17T01:45:25Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a8b29c399636d14713e82c1ffc2b1390601466596343625157a8ff5a6d9809c9","abstract_canon_sha256":"fa5a129b53b1f1a1a2f57a05cbeb7203136ec157082145d93f37063f61047eb8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:15:14.014112Z","signature_b64":"wNwSueFo0EtXPc/iPHGHLOBu7NnCaDOuh9qGwOmenTEwdUoQT88oYGijEc6Ss477nddgS5O84CI5EQMi5A5wBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d07c20e091fb68732c409724684693a4c8c4fe7dca4aec4b070464b77d3b462e","last_reissued_at":"2026-07-05T03:15:14.013707Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:15:14.013707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdaLoss: A computationally-efficient and provably convergent adaptive gradient method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Rachel Ward, Simon Du, Xiaoxia Wu, Yuege Xie","submitted_at":"2021-09-17T01:45:25Z","abstract_excerpt":"We propose a computationally-friendly adaptive learning rate schedule, \"AdaLoss\", which directly uses the information of the loss function to adjust the stepsize in gradient descent methods. We prove that this schedule enjoys linear convergence in linear regression. Moreover, we provide a linear convergence guarantee over the non-convex regime, in the context of two-layer over-parameterized neural networks. If the width of the first-hidden layer in the two-layer networks is sufficiently large (polynomially), then AdaLoss converges robustly \\emph{to the global minimum} in polynomial time. We nu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.08282","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.08282/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.08282","created_at":"2026-07-05T03:15:14.013765+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.08282v1","created_at":"2026-07-05T03:15:14.013765+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.08282","created_at":"2026-07-05T03:15:14.013765+00:00"},{"alias_kind":"pith_short_12","alias_value":"2B6CBYER7NUH","created_at":"2026-07-05T03:15:14.013765+00:00"},{"alias_kind":"pith_short_16","alias_value":"2B6CBYER7NUHGLCA","created_at":"2026-07-05T03:15:14.013765+00:00"},{"alias_kind":"pith_short_8","alias_value":"2B6CBYER","created_at":"2026-07-05T03:15:14.013765+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT","json":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT.json","graph_json":"https://pith.science/api/pith-number/2B6CBYER7NUHGLCAS4SGQRUTUT/graph.json","events_json":"https://pith.science/api/pith-number/2B6CBYER7NUHGLCAS4SGQRUTUT/events.json","paper":"https://pith.science/paper/2B6CBYER"},"agent_actions":{"view_html":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT","download_json":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT.json","view_paper":"https://pith.science/paper/2B6CBYER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.08282&json=true","fetch_graph":"https://pith.science/api/pith-number/2B6CBYER7NUHGLCAS4SGQRUTUT/graph.json","fetch_events":"https://pith.science/api/pith-number/2B6CBYER7NUHGLCAS4SGQRUTUT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT/action/storage_attestation","attest_author":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT/action/author_attestation","sign_citation":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT/action/citation_signature","submit_replication":"https://pith.science/pith/2B6CBYER7NUHGLCAS4SGQRUTUT/action/replication_record"}},"created_at":"2026-07-05T03:15:14.013765+00:00","updated_at":"2026-07-05T03:15:14.013765+00:00"}