{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:RG42RAU4GGGKIGF5LY2PO64IUL","short_pith_number":"pith:RG42RAU4","schema_version":"1.0","canonical_sha256":"89b9a8829c318ca418bd5e34f77b88a2d48ec29ec748c454630045d439d4b568","source":{"kind":"arxiv","id":"1905.12787","version":2},"attestation_state":"computed","paper":{"title":"The Theory Behind Overfitting, Cross Validation, Regularization, Bagging, and Boosting: Tutorial","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Benyamin Ghojogh, Mark Crowley","submitted_at":"2019-05-28T08:18:09Z","abstract_excerpt":"In this tutorial paper, we first define mean squared error, variance, covariance, and bias of both random variables and classification/predictor models. Then, we formulate the true and generalization errors of the model for both training and validation/test instances where we make use of the Stein's Unbiased Risk Estimator (SURE). We define overfitting, underfitting, and generalization using the obtained true and generalization errors. We introduce cross validation and two well-known examples which are $K$-fold and leave-one-out cross validations. We briefly introduce generalized cross validat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1905.12787","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2019-05-28T08:18:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cb3f4ee20b1123638239519acbe89b7eb8ed72cabbaabe080d4ad1c5267b68a5","abstract_canon_sha256":"11cf4acded64ff0061061aa5180e0d55968c78f8b8fb146eedc1e8769eb2f3e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:11:52.236794Z","signature_b64":"sI5qEozkLwf/7BAePOlwnyma9bYL520+J1zOziRYh5u28+Gh4Nf+sRyxLbfO6Xz3n93JcXg3EeX18srncCy+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89b9a8829c318ca418bd5e34f77b88a2d48ec29ec748c454630045d439d4b568","last_reissued_at":"2026-07-05T06:11:52.236367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:11:52.236367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Theory Behind Overfitting, Cross Validation, Regularization, Bagging, and Boosting: Tutorial","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Benyamin Ghojogh, Mark Crowley","submitted_at":"2019-05-28T08:18:09Z","abstract_excerpt":"In this tutorial paper, we first define mean squared error, variance, covariance, and bias of both random variables and classification/predictor models. Then, we formulate the true and generalization errors of the model for both training and validation/test instances where we make use of the Stein's Unbiased Risk Estimator (SURE). We define overfitting, underfitting, and generalization using the obtained true and generalization errors. We introduce cross validation and two well-known examples which are $K$-fold and leave-one-out cross validations. We briefly introduce generalized cross validat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.12787","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1905.12787/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1905.12787","created_at":"2026-07-05T06:11:52.236429+00:00"},{"alias_kind":"arxiv_version","alias_value":"1905.12787v2","created_at":"2026-07-05T06:11:52.236429+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.12787","created_at":"2026-07-05T06:11:52.236429+00:00"},{"alias_kind":"pith_short_12","alias_value":"RG42RAU4GGGK","created_at":"2026-07-05T06:11:52.236429+00:00"},{"alias_kind":"pith_short_16","alias_value":"RG42RAU4GGGKIGF5","created_at":"2026-07-05T06:11:52.236429+00:00"},{"alias_kind":"pith_short_8","alias_value":"RG42RAU4","created_at":"2026-07-05T06:11:52.236429+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22850","citing_title":"To select or not to select: predictively consistent priors instead of model selection","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23090","citing_title":"MAP4TS: A Multi-Aspect Prompting Framework for Time-Series Forecasting with Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08928","citing_title":"A Convolutional Neural Network-Derived Catalog of Solar Flares from Soft X-Ray Observations","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL","json":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL.json","graph_json":"https://pith.science/api/pith-number/RG42RAU4GGGKIGF5LY2PO64IUL/graph.json","events_json":"https://pith.science/api/pith-number/RG42RAU4GGGKIGF5LY2PO64IUL/events.json","paper":"https://pith.science/paper/RG42RAU4"},"agent_actions":{"view_html":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL","download_json":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL.json","view_paper":"https://pith.science/paper/RG42RAU4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1905.12787&json=true","fetch_graph":"https://pith.science/api/pith-number/RG42RAU4GGGKIGF5LY2PO64IUL/graph.json","fetch_events":"https://pith.science/api/pith-number/RG42RAU4GGGKIGF5LY2PO64IUL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL/action/storage_attestation","attest_author":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL/action/author_attestation","sign_citation":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL/action/citation_signature","submit_replication":"https://pith.science/pith/RG42RAU4GGGKIGF5LY2PO64IUL/action/replication_record"}},"created_at":"2026-07-05T06:11:52.236429+00:00","updated_at":"2026-07-05T06:11:52.236429+00:00"}