{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2XCUHGJEFEUQESD6FZNIG7B5A2","short_pith_number":"pith:2XCUHGJE","schema_version":"1.0","canonical_sha256":"d5c5439924292902487e2e5a837c3d06a03ee4db5582c10a7faf24e075af0606","source":{"kind":"arxiv","id":"2505.18346","version":1},"attestation_state":"computed","paper":{"title":"On the Mechanisms of Weak-to-Strong Generalization: A Theoretical Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Behrad Moniri, Hamed Hassani","submitted_at":"2025-05-23T20:09:09Z","abstract_excerpt":"Weak-to-strong generalization, where a student model trained on imperfect labels generated by a weaker teacher nonetheless surpasses that teacher, has been widely observed but the mechanisms that enable it have remained poorly understood. In this paper, through a theoretical analysis of simple models, we uncover three core mechanisms that can drive this phenomenon. First, by analyzing ridge regression, we study the interplay between the teacher and student regularization and prove that a student can compensate for a teacher's under-regularization and achieve lower test error. We also analyze t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18346","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2025-05-23T20:09:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b4418788d04a2ba43b49a1f32c6cb774a7bec39be6d2f56dbcabd0158d464dcf","abstract_canon_sha256":"75d28a190931a09e96d1cf3f136bcefc17bea2498499211fc627646a357a7fe3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:45.179531Z","signature_b64":"lYfbFMMwV0zQs34lBSReNrhmZmCnZeRJQqT1PBNqXRPz6f/sGgNVrBLs7cBuPgFP9X+nAd54uZzdoMA58ZSKBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d5c5439924292902487e2e5a837c3d06a03ee4db5582c10a7faf24e075af0606","last_reissued_at":"2026-07-05T11:08:45.179086Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:45.179086Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Mechanisms of Weak-to-Strong Generalization: A Theoretical Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Behrad Moniri, Hamed Hassani","submitted_at":"2025-05-23T20:09:09Z","abstract_excerpt":"Weak-to-strong generalization, where a student model trained on imperfect labels generated by a weaker teacher nonetheless surpasses that teacher, has been widely observed but the mechanisms that enable it have remained poorly understood. In this paper, through a theoretical analysis of simple models, we uncover three core mechanisms that can drive this phenomenon. First, by analyzing ridge regression, we study the interplay between the teacher and student regularization and prove that a student can compensate for a teacher's under-regularization and achieve lower test error. We also analyze t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18346","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18346/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18346","created_at":"2026-07-05T11:08:45.179143+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18346v1","created_at":"2026-07-05T11:08:45.179143+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18346","created_at":"2026-07-05T11:08:45.179143+00:00"},{"alias_kind":"pith_short_12","alias_value":"2XCUHGJEFEUQ","created_at":"2026-07-05T11:08:45.179143+00:00"},{"alias_kind":"pith_short_16","alias_value":"2XCUHGJEFEUQESD6","created_at":"2026-07-05T11:08:45.179143+00:00"},{"alias_kind":"pith_short_8","alias_value":"2XCUHGJE","created_at":"2026-07-05T11:08:45.179143+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05710","citing_title":"On the Blessing of Pre-training in Weak-to-Strong Generalization","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2","json":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2.json","graph_json":"https://pith.science/api/pith-number/2XCUHGJEFEUQESD6FZNIG7B5A2/graph.json","events_json":"https://pith.science/api/pith-number/2XCUHGJEFEUQESD6FZNIG7B5A2/events.json","paper":"https://pith.science/paper/2XCUHGJE"},"agent_actions":{"view_html":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2","download_json":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2.json","view_paper":"https://pith.science/paper/2XCUHGJE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18346&json=true","fetch_graph":"https://pith.science/api/pith-number/2XCUHGJEFEUQESD6FZNIG7B5A2/graph.json","fetch_events":"https://pith.science/api/pith-number/2XCUHGJEFEUQESD6FZNIG7B5A2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2/action/storage_attestation","attest_author":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2/action/author_attestation","sign_citation":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2/action/citation_signature","submit_replication":"https://pith.science/pith/2XCUHGJEFEUQESD6FZNIG7B5A2/action/replication_record"}},"created_at":"2026-07-05T11:08:45.179143+00:00","updated_at":"2026-07-05T11:08:45.179143+00:00"}