{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MLZJMBUZG2JFJXTPHSHMNVRSNK","short_pith_number":"pith:MLZJMBUZ","schema_version":"1.0","canonical_sha256":"62f2960699369254de6f3c8ec6d6326ab23698f8a67337d4fb8d0b94565095fb","source":{"kind":"arxiv","id":"2301.10921","version":2},"attestation_state":"computed","paper":{"title":"SoftMatch: Addressing the Quantity-Quality Trade-off in Semi-supervised Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Bernt Schiele, Bhiksha Raj, Hao Chen, Jindong Wang, Marios Savvides, Ran Tao, Xing Xie, Yidong Wang, Yue Fan","submitted_at":"2023-01-26T03:53:25Z","abstract_excerpt":"The critical challenge of Semi-Supervised Learning (SSL) is how to effectively leverage the limited labeled data and massive unlabeled data to improve the model's generalization performance. In this paper, we first revisit the popular pseudo-labeling methods via a unified sample weighting formulation and demonstrate the inherent quantity-quality trade-off problem of pseudo-labeling with thresholding, which may prohibit learning. To this end, we propose SoftMatch to overcome the trade-off by maintaining both high quantity and high quality of pseudo-labels during training, effectively exploiting"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.10921","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-01-26T03:53:25Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"564fa9d92e591ad32fc3e74bfe1fdc0031801dcbeb6f31a66f5da6e3081c7a0b","abstract_canon_sha256":"ec384606e2c58f4012215f98e4a57b539b4d79d52dd8009ce543fbd09b7be7ec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:29.216577Z","signature_b64":"beFQi71wmPcKDuYRBEcAOuk4KG3hpWvyCpSh7zyc4HOZ1WM1zrONqg0qLjdxdyrylN01eq9DW2ojrg7Gv02ZAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62f2960699369254de6f3c8ec6d6326ab23698f8a67337d4fb8d0b94565095fb","last_reissued_at":"2026-07-05T05:51:29.216041Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:29.216041Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SoftMatch: Addressing the Quantity-Quality Trade-off in Semi-supervised Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Bernt Schiele, Bhiksha Raj, Hao Chen, Jindong Wang, Marios Savvides, Ran Tao, Xing Xie, Yidong Wang, Yue Fan","submitted_at":"2023-01-26T03:53:25Z","abstract_excerpt":"The critical challenge of Semi-Supervised Learning (SSL) is how to effectively leverage the limited labeled data and massive unlabeled data to improve the model's generalization performance. In this paper, we first revisit the popular pseudo-labeling methods via a unified sample weighting formulation and demonstrate the inherent quantity-quality trade-off problem of pseudo-labeling with thresholding, which may prohibit learning. To this end, we propose SoftMatch to overcome the trade-off by maintaining both high quantity and high quality of pseudo-labels during training, effectively exploiting"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.10921","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.10921/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.10921","created_at":"2026-07-05T05:51:29.216111+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.10921v2","created_at":"2026-07-05T05:51:29.216111+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.10921","created_at":"2026-07-05T05:51:29.216111+00:00"},{"alias_kind":"pith_short_12","alias_value":"MLZJMBUZG2JF","created_at":"2026-07-05T05:51:29.216111+00:00"},{"alias_kind":"pith_short_16","alias_value":"MLZJMBUZG2JFJXTP","created_at":"2026-07-05T05:51:29.216111+00:00"},{"alias_kind":"pith_short_8","alias_value":"MLZJMBUZ","created_at":"2026-07-05T05:51:29.216111+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26973","citing_title":"Geometric Gradient Rectification for Safe Open-Set Semi-Supervised Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00358","citing_title":"PRISM: Prioritized Channel Importance with Semi-supervised Domain Adaptation for Cross-Subject EEG Emotion Recognition","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04516","citing_title":"GeoMin: Data-Efficient Semi-Supervised RLVR via Geometric Distribution Modeling","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30577","citing_title":"APRIL-MedSeg: A Modular Medical Image Segmentation Toolbox Embracing Modern Paradigms","ref_index":226,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30577","citing_title":"APRIL-MedSeg: A Modular Medical Image Segmentation Toolbox Embracing Modern Paradigms","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03993","citing_title":"Can LLMs Learn to Reason Robustly under Noisy Supervision?","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK","json":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK.json","graph_json":"https://pith.science/api/pith-number/MLZJMBUZG2JFJXTPHSHMNVRSNK/graph.json","events_json":"https://pith.science/api/pith-number/MLZJMBUZG2JFJXTPHSHMNVRSNK/events.json","paper":"https://pith.science/paper/MLZJMBUZ"},"agent_actions":{"view_html":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK","download_json":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK.json","view_paper":"https://pith.science/paper/MLZJMBUZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.10921&json=true","fetch_graph":"https://pith.science/api/pith-number/MLZJMBUZG2JFJXTPHSHMNVRSNK/graph.json","fetch_events":"https://pith.science/api/pith-number/MLZJMBUZG2JFJXTPHSHMNVRSNK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK/action/storage_attestation","attest_author":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK/action/author_attestation","sign_citation":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK/action/citation_signature","submit_replication":"https://pith.science/pith/MLZJMBUZG2JFJXTPHSHMNVRSNK/action/replication_record"}},"created_at":"2026-07-05T05:51:29.216111+00:00","updated_at":"2026-07-05T05:51:29.216111+00:00"}