{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KPOFZDJ4OIHWKTQUU2YXRKJPAV","short_pith_number":"pith:KPOFZDJ4","schema_version":"1.0","canonical_sha256":"53dc5c8d3c720f654e14a6b178a92f05472734c6d1ec27f278f29fc3c24e17a6","source":{"kind":"arxiv","id":"2303.00883","version":2},"attestation_state":"computed","paper":{"title":"Variance-reduced Clipping for Non-convex Optimization","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ali Jadbabaie, Amirhossein Reisizadeh, Haochuan Li, Subhro Das","submitted_at":"2023-03-02T00:57:38Z","abstract_excerpt":"Gradient clipping is a standard training technique used in deep learning applications such as large-scale language modeling to mitigate exploding gradients. Recent experimental studies have demonstrated a fairly special behavior in the smoothness of the training objective along its trajectory when trained with gradient clipping. That is, the smoothness grows with the gradient norm. This is in clear contrast to the well-established assumption in folklore non-convex optimization, a.k.a. $L$--smoothness, where the smoothness is assumed to be bounded by a constant $L$ globally. The recently introd"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.00883","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2023-03-02T00:57:38Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"367c50240177c9dd7e53401729549d120bc51f3c41593e8b468f4c0463457ec0","abstract_canon_sha256":"40da724b3d1b00f28caf10ec3724900cfc596d4849699ce0f34abd4ab0b89a7f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:05.206493Z","signature_b64":"JUKdkspjUyGtX6KmQToTtSJ2CPr5TRTKPcs+mp7LkIa97/oraJ0w1aheRts5xfKmZr8UPtQpGHyAlZal5xP5Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53dc5c8d3c720f654e14a6b178a92f05472734c6d1ec27f278f29fc3c24e17a6","last_reissued_at":"2026-07-05T06:17:05.206071Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:05.206071Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Variance-reduced Clipping for Non-convex Optimization","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ali Jadbabaie, Amirhossein Reisizadeh, Haochuan Li, Subhro Das","submitted_at":"2023-03-02T00:57:38Z","abstract_excerpt":"Gradient clipping is a standard training technique used in deep learning applications such as large-scale language modeling to mitigate exploding gradients. Recent experimental studies have demonstrated a fairly special behavior in the smoothness of the training objective along its trajectory when trained with gradient clipping. That is, the smoothness grows with the gradient norm. This is in clear contrast to the well-established assumption in folklore non-convex optimization, a.k.a. $L$--smoothness, where the smoothness is assumed to be bounded by a constant $L$ globally. The recently introd"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.00883","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.00883/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.00883","created_at":"2026-07-05T06:17:05.206119+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.00883v2","created_at":"2026-07-05T06:17:05.206119+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.00883","created_at":"2026-07-05T06:17:05.206119+00:00"},{"alias_kind":"pith_short_12","alias_value":"KPOFZDJ4OIHW","created_at":"2026-07-05T06:17:05.206119+00:00"},{"alias_kind":"pith_short_16","alias_value":"KPOFZDJ4OIHWKTQU","created_at":"2026-07-05T06:17:05.206119+00:00"},{"alias_kind":"pith_short_8","alias_value":"KPOFZDJ4","created_at":"2026-07-05T06:17:05.206119+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.08913","citing_title":"Revisiting Convergence: Shuffling Complexity Beyond Lipschitz Smoothness","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV","json":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV.json","graph_json":"https://pith.science/api/pith-number/KPOFZDJ4OIHWKTQUU2YXRKJPAV/graph.json","events_json":"https://pith.science/api/pith-number/KPOFZDJ4OIHWKTQUU2YXRKJPAV/events.json","paper":"https://pith.science/paper/KPOFZDJ4"},"agent_actions":{"view_html":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV","download_json":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV.json","view_paper":"https://pith.science/paper/KPOFZDJ4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.00883&json=true","fetch_graph":"https://pith.science/api/pith-number/KPOFZDJ4OIHWKTQUU2YXRKJPAV/graph.json","fetch_events":"https://pith.science/api/pith-number/KPOFZDJ4OIHWKTQUU2YXRKJPAV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV/action/storage_attestation","attest_author":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV/action/author_attestation","sign_citation":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV/action/citation_signature","submit_replication":"https://pith.science/pith/KPOFZDJ4OIHWKTQUU2YXRKJPAV/action/replication_record"}},"created_at":"2026-07-05T06:17:05.206119+00:00","updated_at":"2026-07-05T06:17:05.206119+00:00"}