{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:6XQG22ESVX2CZZKSXHH234VIFP","short_pith_number":"pith:6XQG22ES","schema_version":"1.0","canonical_sha256":"f5e06d6892adf42ce552b9cfadf2a82be5ba1a505f3b839177d19b38b36bcaac","source":{"kind":"arxiv","id":"1807.05031","version":6},"attestation_state":"computed","paper":{"title":"On the Relation Between the Sharpest Directions of DNN Loss and the SGD Step Length","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Amos Storkey, Asja Fischer, Nicolas Ballas, Stanis{\\l}aw Jastrz\\k{e}bski, Yoshua Bengio, Zachary Kenton","submitted_at":"2018-07-13T12:17:41Z","abstract_excerpt":"Stochastic Gradient Descent (SGD) based training of neural networks with a large learning rate or a small batch-size typically ends in well-generalizing, flat regions of the weight space, as indicated by small eigenvalues of the Hessian of the training loss. However, the curvature along the SGD trajectory is poorly understood. An empirical investigation shows that initially SGD visits increasingly sharp regions, reaching a maximum sharpness determined by both the learning rate and the batch-size of SGD. When studying the SGD dynamics in relation to the sharpest directions in this initial phase"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1807.05031","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2018-07-13T12:17:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"380136bd379bbc74d79af6accc18702c4b5b8360746b173b3098c12d48fc5049","abstract_canon_sha256":"78d0f41e068a10d503aa40bbc64ab4da84aa6be0b5a7aa8ac3649fa9d561b8ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:27:56.929778Z","signature_b64":"rT6RLrWIQVL7MAqUN/M2Ak85qEdx6tuHkQU/G/BZ4LiZ5UNqK+REttZHKJlqLlaYmeHeaAZMo+kh5//8dHbnAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5e06d6892adf42ce552b9cfadf2a82be5ba1a505f3b839177d19b38b36bcaac","last_reissued_at":"2026-07-05T00:27:56.929284Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:27:56.929284Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Relation Between the Sharpest Directions of DNN Loss and the SGD Step Length","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Amos Storkey, Asja Fischer, Nicolas Ballas, Stanis{\\l}aw Jastrz\\k{e}bski, Yoshua Bengio, Zachary Kenton","submitted_at":"2018-07-13T12:17:41Z","abstract_excerpt":"Stochastic Gradient Descent (SGD) based training of neural networks with a large learning rate or a small batch-size typically ends in well-generalizing, flat regions of the weight space, as indicated by small eigenvalues of the Hessian of the training loss. However, the curvature along the SGD trajectory is poorly understood. An empirical investigation shows that initially SGD visits increasingly sharp regions, reaching a maximum sharpness determined by both the learning rate and the batch-size of SGD. When studying the SGD dynamics in relation to the sharpest directions in this initial phase"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1807.05031","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1807.05031/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1807.05031","created_at":"2026-07-05T00:27:56.929350+00:00"},{"alias_kind":"arxiv_version","alias_value":"1807.05031v6","created_at":"2026-07-05T00:27:56.929350+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1807.05031","created_at":"2026-07-05T00:27:56.929350+00:00"},{"alias_kind":"pith_short_12","alias_value":"6XQG22ESVX2C","created_at":"2026-07-05T00:27:56.929350+00:00"},{"alias_kind":"pith_short_16","alias_value":"6XQG22ESVX2CZZKS","created_at":"2026-07-05T00:27:56.929350+00:00"},{"alias_kind":"pith_short_8","alias_value":"6XQG22ES","created_at":"2026-07-05T00:27:56.929350+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04212","citing_title":"Edge of Stability Selectively Shapes Learning Across the Data Distribution","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30226","citing_title":"Characterizing Optimizer-Dependent Training Dynamics Through Hessian Eigenvector Displacement and Localization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"1907.10732","citing_title":"Hessian based analysis of SGD for Deep Nets: Dynamics and Generalization","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22644","citing_title":"Why SGD is not Brownian Motion: A New Perspective on Stochastic Dynamics","ref_index":222,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16622","citing_title":"Does Weight Decay Enhance Training Stability?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2102.01293","citing_title":"Scaling Laws for Transfer","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2207.05221","citing_title":"Language Models (Mostly) Know What They Know","ref_index":176,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14108","citing_title":"Momentum Further Constrains Sharpness at the Edge of Stochastic Stability","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP","json":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP.json","graph_json":"https://pith.science/api/pith-number/6XQG22ESVX2CZZKSXHH234VIFP/graph.json","events_json":"https://pith.science/api/pith-number/6XQG22ESVX2CZZKSXHH234VIFP/events.json","paper":"https://pith.science/paper/6XQG22ES"},"agent_actions":{"view_html":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP","download_json":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP.json","view_paper":"https://pith.science/paper/6XQG22ES","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1807.05031&json=true","fetch_graph":"https://pith.science/api/pith-number/6XQG22ESVX2CZZKSXHH234VIFP/graph.json","fetch_events":"https://pith.science/api/pith-number/6XQG22ESVX2CZZKSXHH234VIFP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP/action/storage_attestation","attest_author":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP/action/author_attestation","sign_citation":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP/action/citation_signature","submit_replication":"https://pith.science/pith/6XQG22ESVX2CZZKSXHH234VIFP/action/replication_record"}},"created_at":"2026-07-05T00:27:56.929350+00:00","updated_at":"2026-07-05T00:27:56.929350+00:00"}