{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:NCOWY3752XVZTMU5FWIAJNY4S6","short_pith_number":"pith:NCOWY375","schema_version":"1.0","canonical_sha256":"689d6c6ffdd5eb99b29d2d9004b71c97bca76dbbd509715cda8fd2d8666187c3","source":{"kind":"arxiv","id":"2006.04740","version":5},"attestation_state":"computed","paper":{"title":"The Heavy-Tail Phenomenon in SGD","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"math.OC","authors_text":"Lingjiong Zhu, Mert Gurbuzbalaban, Umut \\c{S}im\\c{s}ekli","submitted_at":"2020-06-08T16:43:56Z","abstract_excerpt":"In recent years, various notions of capacity and complexity have been proposed for characterizing the generalization properties of stochastic gradient descent (SGD) in deep learning. Some of the popular notions that correlate well with the performance on unseen data are (i) the `flatness' of the local minimum found by SGD, which is related to the eigenvalues of the Hessian, (ii) the ratio of the stepsize $\\eta$ to the batch-size $b$, which essentially controls the magnitude of the stochastic gradient noise, and (iii) the `tail-index', which measures the heaviness of the tails of the network we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.04740","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2020-06-08T16:43:56Z","cross_cats_sorted":["cs.LG","math.ST","stat.TH"],"title_canon_sha256":"10780a1aa09c6f489cfecf6a2571b6f91bf176829abd2ad49703cab8d2a021d5","abstract_canon_sha256":"ec57055fbd70ec32a61764972fbf7e0175f16993d524af7d48dc0b151e5e61a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:50.863496Z","signature_b64":"lEOeZQ2dL/j2rfezuYUei9xA5lsGE5EWOmxXACOj6Ywz3Cp+s2uSzHbSsv/0py8o9K8kQAw1SKNQNwizq+2LDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"689d6c6ffdd5eb99b29d2d9004b71c97bca76dbbd509715cda8fd2d8666187c3","last_reissued_at":"2026-07-05T02:48:50.863015Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:50.863015Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Heavy-Tail Phenomenon in SGD","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"math.OC","authors_text":"Lingjiong Zhu, Mert Gurbuzbalaban, Umut \\c{S}im\\c{s}ekli","submitted_at":"2020-06-08T16:43:56Z","abstract_excerpt":"In recent years, various notions of capacity and complexity have been proposed for characterizing the generalization properties of stochastic gradient descent (SGD) in deep learning. Some of the popular notions that correlate well with the performance on unseen data are (i) the `flatness' of the local minimum found by SGD, which is related to the eigenvalues of the Hessian, (ii) the ratio of the stepsize $\\eta$ to the batch-size $b$, which essentially controls the magnitude of the stochastic gradient noise, and (iii) the `tail-index', which measures the heaviness of the tails of the network we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.04740","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.04740/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.04740","created_at":"2026-07-05T02:48:50.863073+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.04740v5","created_at":"2026-07-05T02:48:50.863073+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.04740","created_at":"2026-07-05T02:48:50.863073+00:00"},{"alias_kind":"pith_short_12","alias_value":"NCOWY3752XVZ","created_at":"2026-07-05T02:48:50.863073+00:00"},{"alias_kind":"pith_short_16","alias_value":"NCOWY3752XVZTMU5","created_at":"2026-07-05T02:48:50.863073+00:00"},{"alias_kind":"pith_short_8","alias_value":"NCOWY375","created_at":"2026-07-05T02:48:50.863073+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01223","citing_title":"Perspectives on Tsallis Statistics for Artificial Intelligence","ref_index":51,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6","json":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6.json","graph_json":"https://pith.science/api/pith-number/NCOWY3752XVZTMU5FWIAJNY4S6/graph.json","events_json":"https://pith.science/api/pith-number/NCOWY3752XVZTMU5FWIAJNY4S6/events.json","paper":"https://pith.science/paper/NCOWY375"},"agent_actions":{"view_html":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6","download_json":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6.json","view_paper":"https://pith.science/paper/NCOWY375","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.04740&json=true","fetch_graph":"https://pith.science/api/pith-number/NCOWY3752XVZTMU5FWIAJNY4S6/graph.json","fetch_events":"https://pith.science/api/pith-number/NCOWY3752XVZTMU5FWIAJNY4S6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6/action/storage_attestation","attest_author":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6/action/author_attestation","sign_citation":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6/action/citation_signature","submit_replication":"https://pith.science/pith/NCOWY3752XVZTMU5FWIAJNY4S6/action/replication_record"}},"created_at":"2026-07-05T02:48:50.863073+00:00","updated_at":"2026-07-05T02:48:50.863073+00:00"}