{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BLQBDXAFU27IERJWH3AVPHRY3D","short_pith_number":"pith:BLQBDXAF","schema_version":"1.0","canonical_sha256":"0ae011dc05a6be8245363ec1579e38d8c6314f5a102585c6066d670eaf057df8","source":{"kind":"arxiv","id":"2507.12224","version":1},"attestation_state":"computed","paper":{"title":"Optimizers Qualitatively Alter Solutions And We Should Leverage This","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Clare Lyle, Dan Alistarh, Ionut-Vlad Modoranu, James Martens, Naima Elosegui Borras, Petar Velickovic, Razvan Pascanu, Sarath Chandar, Soham De","submitted_at":"2025-07-16T13:33:31Z","abstract_excerpt":"Due to the nonlinear nature of Deep Neural Networks (DNNs), one can not guarantee convergence to a unique global minimum of the loss when using optimizers relying only on local information, such as SGD. Indeed, this was a primary source of skepticism regarding the feasibility of DNNs in the early days of the field. The past decades of progress in deep learning have revealed this skepticism to be misplaced, and a large body of empirical evidence shows that sufficiently large DNNs following standard training protocols exhibit well-behaved optimization dynamics that converge to performant solutio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.12224","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-16T13:33:31Z","cross_cats_sorted":[],"title_canon_sha256":"902d914779e72faca8a636139281afee40e316b877d0daa77b8570d81caa10bf","abstract_canon_sha256":"082b3870d5894970f2c9c7279d2bdca2b4f9754de146ae298dcb625fcce46aa1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:16.964481Z","signature_b64":"Rb2DcExzawJX4sYDzdKcuztIHK+Nc+OAs9Y61yRGWBd55LUwWa1hfwnaOk5xhxMo3F1CPtenDCrsc+JBMTTnDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ae011dc05a6be8245363ec1579e38d8c6314f5a102585c6066d670eaf057df8","last_reissued_at":"2026-07-05T11:38:16.963802Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:16.963802Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimizers Qualitatively Alter Solutions And We Should Leverage This","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Clare Lyle, Dan Alistarh, Ionut-Vlad Modoranu, James Martens, Naima Elosegui Borras, Petar Velickovic, Razvan Pascanu, Sarath Chandar, Soham De","submitted_at":"2025-07-16T13:33:31Z","abstract_excerpt":"Due to the nonlinear nature of Deep Neural Networks (DNNs), one can not guarantee convergence to a unique global minimum of the loss when using optimizers relying only on local information, such as SGD. Indeed, this was a primary source of skepticism regarding the feasibility of DNNs in the early days of the field. The past decades of progress in deep learning have revealed this skepticism to be misplaced, and a large body of empirical evidence shows that sufficiently large DNNs following standard training protocols exhibit well-behaved optimization dynamics that converge to performant solutio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.12224","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.12224/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.12224","created_at":"2026-07-05T11:38:16.963882+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.12224v1","created_at":"2026-07-05T11:38:16.963882+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.12224","created_at":"2026-07-05T11:38:16.963882+00:00"},{"alias_kind":"pith_short_12","alias_value":"BLQBDXAFU27I","created_at":"2026-07-05T11:38:16.963882+00:00"},{"alias_kind":"pith_short_16","alias_value":"BLQBDXAFU27IERJW","created_at":"2026-07-05T11:38:16.963882+00:00"},{"alias_kind":"pith_short_8","alias_value":"BLQBDXAF","created_at":"2026-07-05T11:38:16.963882+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11123","citing_title":"Overcoming Rank Collapse in Feedback Alignment","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27662","citing_title":"How the Optimizer Shapes Learned Solutions in Equivariant Neural Networks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21803","citing_title":"Same Architecture, Different Capacity: Optimizer-Induced Spectral Scaling Laws","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04686","citing_title":"How does the optimizer implicitly bias the model merging loss landscape?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20365","citing_title":"Benefits of Low-Cost Bio-Inspiration in the Age of Overparametrization","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D","json":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D.json","graph_json":"https://pith.science/api/pith-number/BLQBDXAFU27IERJWH3AVPHRY3D/graph.json","events_json":"https://pith.science/api/pith-number/BLQBDXAFU27IERJWH3AVPHRY3D/events.json","paper":"https://pith.science/paper/BLQBDXAF"},"agent_actions":{"view_html":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D","download_json":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D.json","view_paper":"https://pith.science/paper/BLQBDXAF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.12224&json=true","fetch_graph":"https://pith.science/api/pith-number/BLQBDXAFU27IERJWH3AVPHRY3D/graph.json","fetch_events":"https://pith.science/api/pith-number/BLQBDXAFU27IERJWH3AVPHRY3D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D/action/storage_attestation","attest_author":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D/action/author_attestation","sign_citation":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D/action/citation_signature","submit_replication":"https://pith.science/pith/BLQBDXAFU27IERJWH3AVPHRY3D/action/replication_record"}},"created_at":"2026-07-05T11:38:16.963882+00:00","updated_at":"2026-07-05T11:38:16.963882+00:00"}