{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L7AVAQQHYV235H2IWFCRP3CXC3","short_pith_number":"pith:L7AVAQQH","schema_version":"1.0","canonical_sha256":"5fc1504207c575be9f48b14517ec5716f7999c508b06450cfb3f3b7ae3e9abe4","source":{"kind":"arxiv","id":"2501.19105","version":2},"attestation_state":"computed","paper":{"title":"Relating Misfit to Gain in Weak-to-Strong Generalization Beyond the Squared Loss","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.PR"],"primary_cat":"cs.LG","authors_text":"Abhijeet Mulgund, Chirag Pabbaraju","submitted_at":"2025-01-31T12:57:58Z","abstract_excerpt":"The paradigm of weak-to-strong generalization constitutes the training of a strong AI model on data labeled by a weak AI model, with the goal that the strong model nevertheless outperforms its weak supervisor on the target task of interest. For the setting of real-valued regression with the squared loss, recent work quantitatively characterizes the gain in performance of the strong model over the weak model in terms of the misfit between the strong and weak model. We generalize such a characterization to learning tasks whose loss functions correspond to arbitrary Bregman divergences when the s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.19105","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-31T12:57:58Z","cross_cats_sorted":["math.PR"],"title_canon_sha256":"8ccaa808318fbf62289bb5d40b66ceff747222e6a885034b7ff987239a704f03","abstract_canon_sha256":"aed4e2d1bd30af9f397b2c07b45c7e1c802d2a24616d1cbac755f280716463ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:16.037587Z","signature_b64":"wxf/VYh5MwcXAlUbfFl+IeL5EKPYJzlqLs5T2C8jOJ0YZsdTbMUdeBrYsfteKzpbAE0z3/zJJvV6J5L3go/rCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5fc1504207c575be9f48b14517ec5716f7999c508b06450cfb3f3b7ae3e9abe4","last_reissued_at":"2026-07-05T10:09:16.037119Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:16.037119Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Relating Misfit to Gain in Weak-to-Strong Generalization Beyond the Squared Loss","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.PR"],"primary_cat":"cs.LG","authors_text":"Abhijeet Mulgund, Chirag Pabbaraju","submitted_at":"2025-01-31T12:57:58Z","abstract_excerpt":"The paradigm of weak-to-strong generalization constitutes the training of a strong AI model on data labeled by a weak AI model, with the goal that the strong model nevertheless outperforms its weak supervisor on the target task of interest. For the setting of real-valued regression with the squared loss, recent work quantitatively characterizes the gain in performance of the strong model over the weak model in terms of the misfit between the strong and weak model. We generalize such a characterization to learning tasks whose loss functions correspond to arbitrary Bregman divergences when the s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.19105","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.19105/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.19105","created_at":"2026-07-05T10:09:16.037176+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.19105v2","created_at":"2026-07-05T10:09:16.037176+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.19105","created_at":"2026-07-05T10:09:16.037176+00:00"},{"alias_kind":"pith_short_12","alias_value":"L7AVAQQHYV23","created_at":"2026-07-05T10:09:16.037176+00:00"},{"alias_kind":"pith_short_16","alias_value":"L7AVAQQHYV235H2I","created_at":"2026-07-05T10:09:16.037176+00:00"},{"alias_kind":"pith_short_8","alias_value":"L7AVAQQH","created_at":"2026-07-05T10:09:16.037176+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":252,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":252,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05742","citing_title":"Weak-to-Strong Generalization is Nearly Inevitable (in Linear Models)","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05710","citing_title":"On the Blessing of Pre-training in Weak-to-Strong Generalization","ref_index":121,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3","json":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3.json","graph_json":"https://pith.science/api/pith-number/L7AVAQQHYV235H2IWFCRP3CXC3/graph.json","events_json":"https://pith.science/api/pith-number/L7AVAQQHYV235H2IWFCRP3CXC3/events.json","paper":"https://pith.science/paper/L7AVAQQH"},"agent_actions":{"view_html":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3","download_json":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3.json","view_paper":"https://pith.science/paper/L7AVAQQH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.19105&json=true","fetch_graph":"https://pith.science/api/pith-number/L7AVAQQHYV235H2IWFCRP3CXC3/graph.json","fetch_events":"https://pith.science/api/pith-number/L7AVAQQHYV235H2IWFCRP3CXC3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3/action/storage_attestation","attest_author":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3/action/author_attestation","sign_citation":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3/action/citation_signature","submit_replication":"https://pith.science/pith/L7AVAQQHYV235H2IWFCRP3CXC3/action/replication_record"}},"created_at":"2026-07-05T10:09:16.037176+00:00","updated_at":"2026-07-05T10:09:16.037176+00:00"}