{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6ETZASK5FTHL5SYSBNJPRQFKJE","short_pith_number":"pith:6ETZASK5","schema_version":"1.0","canonical_sha256":"f12790495d2ccebecb120b52f8c0aa49301216d47538b122b9f98fe2bfd51257","source":{"kind":"arxiv","id":"2108.12006","version":1},"attestation_state":"computed","paper":{"title":"When and how epochwise double descent happens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cory Stephenson, Tyler Lee","submitted_at":"2021-08-26T19:19:17Z","abstract_excerpt":"Deep neural networks are known to exhibit a `double descent' behavior as the number of parameters increases. Recently, it has also been shown that an `epochwise double descent' effect exists in which the generalization error initially drops, then rises, and finally drops again with increasing training time. This presents a practical problem in that the amount of time required for training is long, and early stopping based on validation performance may result in suboptimal generalization. In this work we develop an analytically tractable model of epochwise double descent that allows us to chara"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.12006","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-08-26T19:19:17Z","cross_cats_sorted":[],"title_canon_sha256":"a75ffad3453f20be9fc1057e0497d83271f1cd18929a71c1baf2644b0324b848","abstract_canon_sha256":"6fe7c62ab1809660d9ee3c66af3a11197b0a161b638ca879c0ff49c059d8196b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:09:20.724439Z","signature_b64":"Bosue5V0Tk9x+pqa4aBVawMnRvzOaYjCyW40I7zIxJ4o3BNzty4FcTT/af3n/C6I/fDN0j1oD/tNVbrT4Q0DAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f12790495d2ccebecb120b52f8c0aa49301216d47538b122b9f98fe2bfd51257","last_reissued_at":"2026-07-05T03:09:20.723987Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:09:20.723987Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When and how epochwise double descent happens","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cory Stephenson, Tyler Lee","submitted_at":"2021-08-26T19:19:17Z","abstract_excerpt":"Deep neural networks are known to exhibit a `double descent' behavior as the number of parameters increases. Recently, it has also been shown that an `epochwise double descent' effect exists in which the generalization error initially drops, then rises, and finally drops again with increasing training time. This presents a practical problem in that the amount of time required for training is long, and early stopping based on validation performance may result in suboptimal generalization. In this work we develop an analytically tractable model of epochwise double descent that allows us to chara"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.12006","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.12006/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.12006","created_at":"2026-07-05T03:09:20.724044+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.12006v1","created_at":"2026-07-05T03:09:20.724044+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.12006","created_at":"2026-07-05T03:09:20.724044+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ETZASK5FTHL","created_at":"2026-07-05T03:09:20.724044+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ETZASK5FTHL5SYS","created_at":"2026-07-05T03:09:20.724044+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ETZASK5","created_at":"2026-07-05T03:09:20.724044+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08350","citing_title":"Grokking and epoch-wise double descent in quantum neural networks","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE","json":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE.json","graph_json":"https://pith.science/api/pith-number/6ETZASK5FTHL5SYSBNJPRQFKJE/graph.json","events_json":"https://pith.science/api/pith-number/6ETZASK5FTHL5SYSBNJPRQFKJE/events.json","paper":"https://pith.science/paper/6ETZASK5"},"agent_actions":{"view_html":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE","download_json":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE.json","view_paper":"https://pith.science/paper/6ETZASK5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.12006&json=true","fetch_graph":"https://pith.science/api/pith-number/6ETZASK5FTHL5SYSBNJPRQFKJE/graph.json","fetch_events":"https://pith.science/api/pith-number/6ETZASK5FTHL5SYSBNJPRQFKJE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE/action/storage_attestation","attest_author":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE/action/author_attestation","sign_citation":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE/action/citation_signature","submit_replication":"https://pith.science/pith/6ETZASK5FTHL5SYSBNJPRQFKJE/action/replication_record"}},"created_at":"2026-07-05T03:09:20.724044+00:00","updated_at":"2026-07-05T03:09:20.724044+00:00"}