{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:EJB3B6LZC4AZPWKKYCFLNY3OGQ","short_pith_number":"pith:EJB3B6LZ","schema_version":"1.0","canonical_sha256":"2243b0f979170197d94ac08ab6e36e342f845e268c4c4460b26993591c0f5afd","source":{"kind":"arxiv","id":"2111.01022","version":2},"attestation_state":"computed","paper":{"title":"Dropout in Training Neural Networks: Flatness of Solution and Noise Structure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Hanxu Zhou, Zhi-Qin John Xu, Zhongwang Zhang","submitted_at":"2021-11-01T15:26:19Z","abstract_excerpt":"It is important to understand how the popular regularization method dropout helps the neural network training find a good generalization solution. In this work, we show that the training with dropout finds the neural network with a flatter minimum compared with standard gradient descent training. We further find that the variance of a noise induced by the dropout is larger at the sharper direction of the loss landscape and the Hessian of the loss landscape at the found minima aligns with the noise covariance matrix by experiments on various datasets, i.e., MNIST, CIFAR-10, CIFAR-100 and Multi3"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.01022","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-11-01T15:26:19Z","cross_cats_sorted":[],"title_canon_sha256":"a9f75c95e7a46da40fb15aad5a5882f73aa6077d201683a4a726c207fbf2e653","abstract_canon_sha256":"fd01c273dbe51b35fece7223cff55ec3584d995ec13663cf84739fe9a55c6c1b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:25:13.254724Z","signature_b64":"4gu6jQ9fuUp9IRCdUrMdkiFG2sX8etW1uRZeBYwchHJySgsXMVQk2u+QFrRCiGsyqH+Y5eTfZ5L5ejdHeWHZDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2243b0f979170197d94ac08ab6e36e342f845e268c4c4460b26993591c0f5afd","last_reissued_at":"2026-07-05T04:25:13.254236Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:25:13.254236Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dropout in Training Neural Networks: Flatness of Solution and Noise Structure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Hanxu Zhou, Zhi-Qin John Xu, Zhongwang Zhang","submitted_at":"2021-11-01T15:26:19Z","abstract_excerpt":"It is important to understand how the popular regularization method dropout helps the neural network training find a good generalization solution. In this work, we show that the training with dropout finds the neural network with a flatter minimum compared with standard gradient descent training. We further find that the variance of a noise induced by the dropout is larger at the sharper direction of the loss landscape and the Hessian of the loss landscape at the found minima aligns with the noise covariance matrix by experiments on various datasets, i.e., MNIST, CIFAR-10, CIFAR-100 and Multi3"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.01022","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.01022/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.01022","created_at":"2026-07-05T04:25:13.254290+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.01022v2","created_at":"2026-07-05T04:25:13.254290+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.01022","created_at":"2026-07-05T04:25:13.254290+00:00"},{"alias_kind":"pith_short_12","alias_value":"EJB3B6LZC4AZ","created_at":"2026-07-05T04:25:13.254290+00:00"},{"alias_kind":"pith_short_16","alias_value":"EJB3B6LZC4AZPWKK","created_at":"2026-07-05T04:25:13.254290+00:00"},{"alias_kind":"pith_short_8","alias_value":"EJB3B6LZ","created_at":"2026-07-05T04:25:13.254290+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.03031","citing_title":"On the Mathematical Impossibility of Safe Universal Approximators","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ","json":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ.json","graph_json":"https://pith.science/api/pith-number/EJB3B6LZC4AZPWKKYCFLNY3OGQ/graph.json","events_json":"https://pith.science/api/pith-number/EJB3B6LZC4AZPWKKYCFLNY3OGQ/events.json","paper":"https://pith.science/paper/EJB3B6LZ"},"agent_actions":{"view_html":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ","download_json":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ.json","view_paper":"https://pith.science/paper/EJB3B6LZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.01022&json=true","fetch_graph":"https://pith.science/api/pith-number/EJB3B6LZC4AZPWKKYCFLNY3OGQ/graph.json","fetch_events":"https://pith.science/api/pith-number/EJB3B6LZC4AZPWKKYCFLNY3OGQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ/action/storage_attestation","attest_author":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ/action/author_attestation","sign_citation":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ/action/citation_signature","submit_replication":"https://pith.science/pith/EJB3B6LZC4AZPWKKYCFLNY3OGQ/action/replication_record"}},"created_at":"2026-07-05T04:25:13.254290+00:00","updated_at":"2026-07-05T04:25:13.254290+00:00"}