{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:KEFRZNSXGKOMLK3BZK4AYUGPJ7","short_pith_number":"pith:KEFRZNSX","schema_version":"1.0","canonical_sha256":"510b1cb657329cc5ab61cab80c50cf4fe925df106e5a4031fcec3e0b8a51ec14","source":{"kind":"arxiv","id":"2205.10697","version":6},"attestation_state":"computed","paper":{"title":"Lassoed Tree Boosting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Alejandro Schuler, Mark van der Laan, Yi Li","submitted_at":"2022-05-22T00:34:41Z","abstract_excerpt":"Gradient boosting performs exceptionally in most prediction problems and scales well to large datasets. In this paper we prove that a ``lassoed'' gradient boosted tree algorithm with early stopping achieves faster than $n^{-1/4}$ L2 convergence in the large nonparametric space of cadlag functions of bounded sectional variation. This rate is remarkable because it does not depend on the dimension, sparsity, or smoothness. We use simulation and real data to confirm our theory and demonstrate empirical performance and scalability on par with standard boosting. Our convergence proofs are based on a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.10697","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2022-05-22T00:34:41Z","cross_cats_sorted":["cs.LG","math.ST","stat.TH"],"title_canon_sha256":"bb8465b896e8db0d9bb18022f521f3df40bee258b32b95bc1eb8e4cf260d2bcf","abstract_canon_sha256":"2b5893e5370f75571ab9e46ac4148ed9f85905c624e69a03239879153552a3aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:21:59.508450Z","signature_b64":"bA7FvHlxSj3iZWtIyPUieQ693eNxNPN9skt6a9Hgtv+sx+YhmAlh1gIwgpZaJEHqRdrCU7SKWLGaHGDL5n2qDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"510b1cb657329cc5ab61cab80c50cf4fe925df106e5a4031fcec3e0b8a51ec14","last_reissued_at":"2026-07-05T07:21:59.507992Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:21:59.507992Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lassoed Tree Boosting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Alejandro Schuler, Mark van der Laan, Yi Li","submitted_at":"2022-05-22T00:34:41Z","abstract_excerpt":"Gradient boosting performs exceptionally in most prediction problems and scales well to large datasets. In this paper we prove that a ``lassoed'' gradient boosted tree algorithm with early stopping achieves faster than $n^{-1/4}$ L2 convergence in the large nonparametric space of cadlag functions of bounded sectional variation. This rate is remarkable because it does not depend on the dimension, sparsity, or smoothness. We use simulation and real data to confirm our theory and demonstrate empirical performance and scalability on par with standard boosting. Our convergence proofs are based on a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.10697","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.10697/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.10697","created_at":"2026-07-05T07:21:59.508050+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.10697v6","created_at":"2026-07-05T07:21:59.508050+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.10697","created_at":"2026-07-05T07:21:59.508050+00:00"},{"alias_kind":"pith_short_12","alias_value":"KEFRZNSXGKOM","created_at":"2026-07-05T07:21:59.508050+00:00"},{"alias_kind":"pith_short_16","alias_value":"KEFRZNSXGKOMLK3B","created_at":"2026-07-05T07:21:59.508050+00:00"},{"alias_kind":"pith_short_8","alias_value":"KEFRZNSX","created_at":"2026-07-05T07:21:59.508050+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15483","citing_title":"Improving the Efficiency of Subgroup Analysis in Randomized Controlled Trials with TMLE","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7","json":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7.json","graph_json":"https://pith.science/api/pith-number/KEFRZNSXGKOMLK3BZK4AYUGPJ7/graph.json","events_json":"https://pith.science/api/pith-number/KEFRZNSXGKOMLK3BZK4AYUGPJ7/events.json","paper":"https://pith.science/paper/KEFRZNSX"},"agent_actions":{"view_html":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7","download_json":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7.json","view_paper":"https://pith.science/paper/KEFRZNSX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.10697&json=true","fetch_graph":"https://pith.science/api/pith-number/KEFRZNSXGKOMLK3BZK4AYUGPJ7/graph.json","fetch_events":"https://pith.science/api/pith-number/KEFRZNSXGKOMLK3BZK4AYUGPJ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7/action/storage_attestation","attest_author":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7/action/author_attestation","sign_citation":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7/action/citation_signature","submit_replication":"https://pith.science/pith/KEFRZNSXGKOMLK3BZK4AYUGPJ7/action/replication_record"}},"created_at":"2026-07-05T07:21:59.508050+00:00","updated_at":"2026-07-05T07:21:59.508050+00:00"}