{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:I3EZRICWN64U3DCKWX2VRDFHJQ","short_pith_number":"pith:I3EZRICW","schema_version":"1.0","canonical_sha256":"46c998a0566fb94d8c4ab5f5588ca74c1dfb57a6d83e83cc793a932a688a93f1","source":{"kind":"arxiv","id":"2010.08127","version":2},"attestation_state":"computed","paper":{"title":"The Deep Bootstrap Framework: Good Online Learners are Good Offline Generalizers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.NE","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Behnam Neyshabur, Hanie Sedghi, Preetum Nakkiran","submitted_at":"2020-10-16T03:07:49Z","abstract_excerpt":"We propose a new framework for reasoning about generalization in deep learning. The core idea is to couple the Real World, where optimizers take stochastic gradient steps on the empirical loss, to an Ideal World, where optimizers take steps on the population loss. This leads to an alternate decomposition of test error into: (1) the Ideal World test error plus (2) the gap between the two worlds. If the gap (2) is universally small, this reduces the problem of generalization in offline learning to the problem of optimization in online learning. We then give empirical evidence that this gap betwe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.08127","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-16T03:07:49Z","cross_cats_sorted":["cs.CV","cs.NE","math.ST","stat.ML","stat.TH"],"title_canon_sha256":"3cecb284cd3ef5c08c1ed2ea8dc49680c34b3e137df5a93f73895ed99744faf6","abstract_canon_sha256":"3afcbbec1cb5cd62535e9eca28a040d9a21aa0525079cd1183725b67015eebcd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:16:27.791349Z","signature_b64":"Yc5wd4UZRR2uQnySXRbeXNn8amOPzvUlHula+1WJrBXnDWRpNsfHlUFJUdip5NJgzGXglDeePvrhlZcXTvnKCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"46c998a0566fb94d8c4ab5f5588ca74c1dfb57a6d83e83cc793a932a688a93f1","last_reissued_at":"2026-07-05T02:16:27.790989Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:16:27.790989Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Deep Bootstrap Framework: Good Online Learners are Good Offline Generalizers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.NE","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Behnam Neyshabur, Hanie Sedghi, Preetum Nakkiran","submitted_at":"2020-10-16T03:07:49Z","abstract_excerpt":"We propose a new framework for reasoning about generalization in deep learning. The core idea is to couple the Real World, where optimizers take stochastic gradient steps on the empirical loss, to an Ideal World, where optimizers take steps on the population loss. This leads to an alternate decomposition of test error into: (1) the Ideal World test error plus (2) the gap between the two worlds. If the gap (2) is universally small, this reduces the problem of generalization in offline learning to the problem of optimization in online learning. We then give empirical evidence that this gap betwe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.08127","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.08127/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.08127","created_at":"2026-07-05T02:16:27.791042+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.08127v2","created_at":"2026-07-05T02:16:27.791042+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.08127","created_at":"2026-07-05T02:16:27.791042+00:00"},{"alias_kind":"pith_short_12","alias_value":"I3EZRICWN64U","created_at":"2026-07-05T02:16:27.791042+00:00"},{"alias_kind":"pith_short_16","alias_value":"I3EZRICWN64U3DCK","created_at":"2026-07-05T02:16:27.791042+00:00"},{"alias_kind":"pith_short_8","alias_value":"I3EZRICW","created_at":"2026-07-05T02:16:27.791042+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.04774","citing_title":"Theory of Optimal Learning Rate Schedules and Scaling Laws for a Random Feature Model","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12410","citing_title":"Model-based Bootstrap of Controlled Markov Chains","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05115","citing_title":"Manifold Steering Reveals the Shared Geometry of Neural Network Representation and Behavior","ref_index":217,"is_internal_anchor":false},{"citing_arxiv_id":"2206.04615","citing_title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ","json":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ.json","graph_json":"https://pith.science/api/pith-number/I3EZRICWN64U3DCKWX2VRDFHJQ/graph.json","events_json":"https://pith.science/api/pith-number/I3EZRICWN64U3DCKWX2VRDFHJQ/events.json","paper":"https://pith.science/paper/I3EZRICW"},"agent_actions":{"view_html":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ","download_json":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ.json","view_paper":"https://pith.science/paper/I3EZRICW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.08127&json=true","fetch_graph":"https://pith.science/api/pith-number/I3EZRICWN64U3DCKWX2VRDFHJQ/graph.json","fetch_events":"https://pith.science/api/pith-number/I3EZRICWN64U3DCKWX2VRDFHJQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ/action/storage_attestation","attest_author":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ/action/author_attestation","sign_citation":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ/action/citation_signature","submit_replication":"https://pith.science/pith/I3EZRICWN64U3DCKWX2VRDFHJQ/action/replication_record"}},"created_at":"2026-07-05T02:16:27.791042+00:00","updated_at":"2026-07-05T02:16:27.791042+00:00"}