{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IANF3QSIJFVRY6DZ6EBB5XNRSR","short_pith_number":"pith:IANF3QSI","schema_version":"1.0","canonical_sha256":"401a5dc248496b1c7879f1021eddb194491a2813cc15c75d3f2365732bc2a40c","source":{"kind":"arxiv","id":"2505.24788","version":1},"attestation_state":"computed","paper":{"title":"Drop Dropout on Single-Epoch Language Model Pretraining","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Christopher D. Manning, Houjun Liu, John Bauer","submitted_at":"2025-05-30T16:48:38Z","abstract_excerpt":"Originally, dropout was seen as a breakthrough regularization technique that reduced overfitting and improved performance in almost all applications of deep learning by reducing overfitting. Yet, single-epoch pretraining tasks common to modern LLMs yield minimal overfitting, leading to dropout not being used for large LLMs. Nevertheless, no thorough empirical investigation has been done on the role of dropout in LM pretraining. Through experiments in single-epoch pretraining of both masked (BERT) and autoregressive (Pythia 160M and 1.4B) LMs with varying levels of dropout, we find that downstr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24788","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-30T16:48:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"37978a0e07d4b637bd7cc622d71bbc650e5b1463cf4aa0af6f5351d1aebfcd20","abstract_canon_sha256":"03fd3e35f8b0ade1e55ca4bfa748f81349c5461fd91ed71a354704c00483a8df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:56.706665Z","signature_b64":"ZFkbEYj3VAqwirgCVlvcfTGNK06R7nM9ZMNIb0WhXuXtjM637c5TrVmVm3ocZi/KTbf9T+H3ZdfCb12BKsKzDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"401a5dc248496b1c7879f1021eddb194491a2813cc15c75d3f2365732bc2a40c","last_reissued_at":"2026-07-05T11:12:56.706157Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:56.706157Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Drop Dropout on Single-Epoch Language Model Pretraining","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Christopher D. Manning, Houjun Liu, John Bauer","submitted_at":"2025-05-30T16:48:38Z","abstract_excerpt":"Originally, dropout was seen as a breakthrough regularization technique that reduced overfitting and improved performance in almost all applications of deep learning by reducing overfitting. Yet, single-epoch pretraining tasks common to modern LLMs yield minimal overfitting, leading to dropout not being used for large LLMs. Nevertheless, no thorough empirical investigation has been done on the role of dropout in LM pretraining. Through experiments in single-epoch pretraining of both masked (BERT) and autoregressive (Pythia 160M and 1.4B) LMs with varying levels of dropout, we find that downstr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24788","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24788/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24788","created_at":"2026-07-05T11:12:56.706222+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24788v1","created_at":"2026-07-05T11:12:56.706222+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24788","created_at":"2026-07-05T11:12:56.706222+00:00"},{"alias_kind":"pith_short_12","alias_value":"IANF3QSIJFVR","created_at":"2026-07-05T11:12:56.706222+00:00"},{"alias_kind":"pith_short_16","alias_value":"IANF3QSIJFVRY6DZ","created_at":"2026-07-05T11:12:56.706222+00:00"},{"alias_kind":"pith_short_8","alias_value":"IANF3QSI","created_at":"2026-07-05T11:12:56.706222+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17465","citing_title":"Language models recognize dropout and Gaussian noise applied to their activations","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR","json":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR.json","graph_json":"https://pith.science/api/pith-number/IANF3QSIJFVRY6DZ6EBB5XNRSR/graph.json","events_json":"https://pith.science/api/pith-number/IANF3QSIJFVRY6DZ6EBB5XNRSR/events.json","paper":"https://pith.science/paper/IANF3QSI"},"agent_actions":{"view_html":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR","download_json":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR.json","view_paper":"https://pith.science/paper/IANF3QSI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24788&json=true","fetch_graph":"https://pith.science/api/pith-number/IANF3QSIJFVRY6DZ6EBB5XNRSR/graph.json","fetch_events":"https://pith.science/api/pith-number/IANF3QSIJFVRY6DZ6EBB5XNRSR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR/action/storage_attestation","attest_author":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR/action/author_attestation","sign_citation":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR/action/citation_signature","submit_replication":"https://pith.science/pith/IANF3QSIJFVRY6DZ6EBB5XNRSR/action/replication_record"}},"created_at":"2026-07-05T11:12:56.706222+00:00","updated_at":"2026-07-05T11:12:56.706222+00:00"}