{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:BSCGEISOSQ2IPYFN47TOA5Z374","short_pith_number":"pith:BSCGEISO","schema_version":"1.0","canonical_sha256":"0c8462224e943487e0ade7e6e0773bff3c9b83bc67185c4e5f6ecdb72d437d47","source":{"kind":"arxiv","id":"2204.11326","version":3},"attestation_state":"computed","paper":{"title":"Beyond the Quadratic Approximation: the Multiscale Structure of Neural Network Loss Landscapes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chao Ma, Daniel Kunin, Lei Wu, Lexing Ying","submitted_at":"2022-04-24T17:34:12Z","abstract_excerpt":"A quadratic approximation of neural network loss landscapes has been extensively used to study the optimization process of these networks. Though, it usually holds in a very small neighborhood of the minimum, it cannot explain many phenomena observed during the optimization process. In this work, we study the structure of neural network loss functions and its implication on optimization in a region beyond the reach of a good quadratic approximation. Numerically, we observe that neural network loss functions possesses a multiscale structure, manifested in two ways: (1) in a neighborhood of mini"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.11326","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-24T17:34:12Z","cross_cats_sorted":[],"title_canon_sha256":"3710aaf3bd53515b73e61896df8e204fb69bbdd10db00517d19cc78970a31617","abstract_canon_sha256":"e59ae543c839fb863d66be1eedb2ad2de676b1e2796a9e2066b8f2386c2eccb2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:34:04.415664Z","signature_b64":"zli6SviOiMQXq1fHojvjQqVqHC7imuhVSF5yPSJxQzsgZXtXZTMqPLwLzNlseCs2j38g9Ac3R4A0+7unBaBcCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c8462224e943487e0ade7e6e0773bff3c9b83bc67185c4e5f6ecdb72d437d47","last_reissued_at":"2026-07-05T04:34:04.415152Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:34:04.415152Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond the Quadratic Approximation: the Multiscale Structure of Neural Network Loss Landscapes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chao Ma, Daniel Kunin, Lei Wu, Lexing Ying","submitted_at":"2022-04-24T17:34:12Z","abstract_excerpt":"A quadratic approximation of neural network loss landscapes has been extensively used to study the optimization process of these networks. Though, it usually holds in a very small neighborhood of the minimum, it cannot explain many phenomena observed during the optimization process. In this work, we study the structure of neural network loss functions and its implication on optimization in a region beyond the reach of a good quadratic approximation. Numerically, we observe that neural network loss functions possesses a multiscale structure, manifested in two ways: (1) in a neighborhood of mini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.11326","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.11326/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.11326","created_at":"2026-07-05T04:34:04.415225+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.11326v3","created_at":"2026-07-05T04:34:04.415225+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.11326","created_at":"2026-07-05T04:34:04.415225+00:00"},{"alias_kind":"pith_short_12","alias_value":"BSCGEISOSQ2I","created_at":"2026-07-05T04:34:04.415225+00:00"},{"alias_kind":"pith_short_16","alias_value":"BSCGEISOSQ2IPYFN","created_at":"2026-07-05T04:34:04.415225+00:00"},{"alias_kind":"pith_short_8","alias_value":"BSCGEISO","created_at":"2026-07-05T04:34:04.415225+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09012","citing_title":"Understanding Quantization-Aware Training: Gradients at Quantized Weights Bias to the Low-Loss Basin","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05372","citing_title":"From Nonsmooth Minima to Smooth Branches via Heat Kernel Regularization","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374","json":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374.json","graph_json":"https://pith.science/api/pith-number/BSCGEISOSQ2IPYFN47TOA5Z374/graph.json","events_json":"https://pith.science/api/pith-number/BSCGEISOSQ2IPYFN47TOA5Z374/events.json","paper":"https://pith.science/paper/BSCGEISO"},"agent_actions":{"view_html":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374","download_json":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374.json","view_paper":"https://pith.science/paper/BSCGEISO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.11326&json=true","fetch_graph":"https://pith.science/api/pith-number/BSCGEISOSQ2IPYFN47TOA5Z374/graph.json","fetch_events":"https://pith.science/api/pith-number/BSCGEISOSQ2IPYFN47TOA5Z374/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374/action/storage_attestation","attest_author":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374/action/author_attestation","sign_citation":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374/action/citation_signature","submit_replication":"https://pith.science/pith/BSCGEISOSQ2IPYFN47TOA5Z374/action/replication_record"}},"created_at":"2026-07-05T04:34:04.415225+00:00","updated_at":"2026-07-05T04:34:04.415225+00:00"}