{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:HDNHLG5DCXRCNGRCKRSC7BUYXK","short_pith_number":"pith:HDNHLG5D","schema_version":"1.0","canonical_sha256":"38da759ba315e2269a2254642f8698baaf7658299daa585d3179f0fda60e3cf6","source":{"kind":"arxiv","id":"2205.07739","version":3},"attestation_state":"computed","paper":{"title":"The Role of Pseudo-labels in Self-training Linear Classifiers on High-dimensional Gaussian Mixture Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cond-mat.stat-mech","cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Takashi Takahashi","submitted_at":"2022-05-16T15:02:44Z","abstract_excerpt":"Self-training (ST) is a simple yet effective semi-supervised learning method. However, why and how ST improves generalization performance by using potentially erroneous pseudo-labels is still not well understood. To deepen the understanding of ST, we derive and analyze a sharp characterization of the behavior of iterative ST when training a linear classifier by minimizing the ridge-regularized convex loss on binary Gaussian mixtures, in the asymptotic limit where input dimension and data size diverge proportionally. The results show that ST improves generalization in different ways depending o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.07739","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2022-05-16T15:02:44Z","cross_cats_sorted":["cond-mat.dis-nn","cond-mat.stat-mech","cs.LG","math.ST","stat.TH"],"title_canon_sha256":"4584c5b53fe916e4d8889e467ce10afad131e65a45ee9d2e8b7fe4bb092596bc","abstract_canon_sha256":"801cbecc92b740c6d478210c0c26bb5505da54c6bc239483de1128335083e0aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:16:18.719574Z","signature_b64":"+l17CadW15FIv9B+0Ib+z/xX/5IL2aoQRupuL0+49vZKL2qvh+/NFKOdWb0ifonCHlIUqgOlOqQ0XGWtyR8bCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38da759ba315e2269a2254642f8698baaf7658299daa585d3179f0fda60e3cf6","last_reissued_at":"2026-07-05T08:16:18.719075Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:16:18.719075Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Role of Pseudo-labels in Self-training Linear Classifiers on High-dimensional Gaussian Mixture Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cond-mat.stat-mech","cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Takashi Takahashi","submitted_at":"2022-05-16T15:02:44Z","abstract_excerpt":"Self-training (ST) is a simple yet effective semi-supervised learning method. However, why and how ST improves generalization performance by using potentially erroneous pseudo-labels is still not well understood. To deepen the understanding of ST, we derive and analyze a sharp characterization of the behavior of iterative ST when training a linear classifier by minimizing the ridge-regularized convex loss on binary Gaussian mixtures, in the asymptotic limit where input dimension and data size diverge proportionally. The results show that ST improves generalization in different ways depending o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.07739","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.07739/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.07739","created_at":"2026-07-05T08:16:18.719159+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.07739v3","created_at":"2026-07-05T08:16:18.719159+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.07739","created_at":"2026-07-05T08:16:18.719159+00:00"},{"alias_kind":"pith_short_12","alias_value":"HDNHLG5DCXRC","created_at":"2026-07-05T08:16:18.719159+00:00"},{"alias_kind":"pith_short_16","alias_value":"HDNHLG5DCXRCNGRC","created_at":"2026-07-05T08:16:18.719159+00:00"},{"alias_kind":"pith_short_8","alias_value":"HDNHLG5D","created_at":"2026-07-05T08:16:18.719159+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK","json":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK.json","graph_json":"https://pith.science/api/pith-number/HDNHLG5DCXRCNGRCKRSC7BUYXK/graph.json","events_json":"https://pith.science/api/pith-number/HDNHLG5DCXRCNGRCKRSC7BUYXK/events.json","paper":"https://pith.science/paper/HDNHLG5D"},"agent_actions":{"view_html":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK","download_json":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK.json","view_paper":"https://pith.science/paper/HDNHLG5D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.07739&json=true","fetch_graph":"https://pith.science/api/pith-number/HDNHLG5DCXRCNGRCKRSC7BUYXK/graph.json","fetch_events":"https://pith.science/api/pith-number/HDNHLG5DCXRCNGRCKRSC7BUYXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK/action/storage_attestation","attest_author":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK/action/author_attestation","sign_citation":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK/action/citation_signature","submit_replication":"https://pith.science/pith/HDNHLG5DCXRCNGRCKRSC7BUYXK/action/replication_record"}},"created_at":"2026-07-05T08:16:18.719159+00:00","updated_at":"2026-07-05T08:16:18.719159+00:00"}