{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HLGQYN7DHBT524JNRGZ35CT2M2","short_pith_number":"pith:HLGQYN7D","schema_version":"1.0","canonical_sha256":"3acd0c37e33867dd712d89b3be8a7a66858eb8991255c47f770d0252d900fd47","source":{"kind":"arxiv","id":"2501.14458","version":1},"attestation_state":"computed","paper":{"title":"A Survey of Optimization Methods for Training DL Models: Theoretical Perspective on Convergence and Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","math.OC"],"primary_cat":"cs.LG","authors_text":"Anna Choromanska, Jing Wang","submitted_at":"2025-01-24T12:42:38Z","abstract_excerpt":"As data sets grow in size and complexity, it is becoming more difficult to pull useful features from them using hand-crafted feature extractors. For this reason, deep learning (DL) frameworks are now widely popular. The Holy Grail of DL and one of the most mysterious challenges in all of modern ML is to develop a fundamental understanding of DL optimization and generalization. While numerous optimization techniques have been introduced in the literature to navigate the exploration of the highly non-convex DL optimization landscape, many survey papers reviewing them primarily focus on summarizi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14458","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-24T12:42:38Z","cross_cats_sorted":["cs.DC","math.OC"],"title_canon_sha256":"53fec53d4ddeea4143fdd15658310dda6a5033b67f2fbc21a2064025cf8d79e4","abstract_canon_sha256":"87b5e6dafe151a5badd148c52fc508e439598c34e2e76ae3e2a059cbe1792a58"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:01.110648Z","signature_b64":"Gft/0p12dGZ8rfgVSjvJppWzlWVp7LTtKh0dFCpDuDBdI4ESZj36IrQjFkRAwVN3G0o6XWzvA1yRgQS6+pwRBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3acd0c37e33867dd712d89b3be8a7a66858eb8991255c47f770d0252d900fd47","last_reissued_at":"2026-07-05T10:05:01.110152Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:01.110152Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Optimization Methods for Training DL Models: Theoretical Perspective on Convergence and Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","math.OC"],"primary_cat":"cs.LG","authors_text":"Anna Choromanska, Jing Wang","submitted_at":"2025-01-24T12:42:38Z","abstract_excerpt":"As data sets grow in size and complexity, it is becoming more difficult to pull useful features from them using hand-crafted feature extractors. For this reason, deep learning (DL) frameworks are now widely popular. The Holy Grail of DL and one of the most mysterious challenges in all of modern ML is to develop a fundamental understanding of DL optimization and generalization. While numerous optimization techniques have been introduced in the literature to navigate the exploration of the highly non-convex DL optimization landscape, many survey papers reviewing them primarily focus on summarizi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14458","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14458/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14458","created_at":"2026-07-05T10:05:01.110216+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14458v1","created_at":"2026-07-05T10:05:01.110216+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14458","created_at":"2026-07-05T10:05:01.110216+00:00"},{"alias_kind":"pith_short_12","alias_value":"HLGQYN7DHBT5","created_at":"2026-07-05T10:05:01.110216+00:00"},{"alias_kind":"pith_short_16","alias_value":"HLGQYN7DHBT524JN","created_at":"2026-07-05T10:05:01.110216+00:00"},{"alias_kind":"pith_short_8","alias_value":"HLGQYN7D","created_at":"2026-07-05T10:05:01.110216+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.13196","citing_title":"A Physics-Inspired Optimizer: Velocity Regularized Adam","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2","json":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2.json","graph_json":"https://pith.science/api/pith-number/HLGQYN7DHBT524JNRGZ35CT2M2/graph.json","events_json":"https://pith.science/api/pith-number/HLGQYN7DHBT524JNRGZ35CT2M2/events.json","paper":"https://pith.science/paper/HLGQYN7D"},"agent_actions":{"view_html":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2","download_json":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2.json","view_paper":"https://pith.science/paper/HLGQYN7D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14458&json=true","fetch_graph":"https://pith.science/api/pith-number/HLGQYN7DHBT524JNRGZ35CT2M2/graph.json","fetch_events":"https://pith.science/api/pith-number/HLGQYN7DHBT524JNRGZ35CT2M2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2/action/storage_attestation","attest_author":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2/action/author_attestation","sign_citation":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2/action/citation_signature","submit_replication":"https://pith.science/pith/HLGQYN7DHBT524JNRGZ35CT2M2/action/replication_record"}},"created_at":"2026-07-05T10:05:01.110216+00:00","updated_at":"2026-07-05T10:05:01.110216+00:00"}