{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L4LPLZZC5ZSY2SBKIGBEO5FUDK","short_pith_number":"pith:L4LPLZZC","schema_version":"1.0","canonical_sha256":"5f16f5e722ee658d482a41824774b41a81114b9a2b2b63bdba6e95d6b692ca55","source":{"kind":"arxiv","id":"2502.10119","version":1},"attestation_state":"computed","paper":{"title":"SeWA: Selective Weight Average via Probabilistic Masking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Dianhai Yu, Guoxia Wang, Li Shen, Peng Wang, Quan Zheng, Shengchao Hu, Zerui Tao","submitted_at":"2025-02-14T12:35:21Z","abstract_excerpt":"Weight averaging has become a standard technique for enhancing model performance. However, methods such as Stochastic Weight Averaging (SWA) and Latest Weight Averaging (LAWA) often require manually designed procedures to sample from the training trajectory, and the results depend heavily on hyperparameter tuning. To minimize human effort, this paper proposes a simple yet efficient algorithm called Selective Weight Averaging (SeWA), which adaptively selects checkpoints during the final stages of training for averaging. Based on SeWA, we show that only a few points are needed to achieve better "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.10119","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-14T12:35:21Z","cross_cats_sorted":[],"title_canon_sha256":"7303fa4e6d7ee80fb5ca6a141d3a366f2454c790b3a144e9049769236655c1a0","abstract_canon_sha256":"c006312c6a82ab4f24a600d84a811cd49935e5990c4b47bb79cabb6292e50e0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:28.500758Z","signature_b64":"agTYALchzC8pyUPAfvjIraYqY2ErqM1psaMuRU4mt1q9rQnx1Ep+sO7JsfU3IWh1ivNBAiJYcn1AQ698TEDHAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f16f5e722ee658d482a41824774b41a81114b9a2b2b63bdba6e95d6b692ca55","last_reissued_at":"2026-07-05T10:14:28.500232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:28.500232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SeWA: Selective Weight Average via Probabilistic Masking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Dianhai Yu, Guoxia Wang, Li Shen, Peng Wang, Quan Zheng, Shengchao Hu, Zerui Tao","submitted_at":"2025-02-14T12:35:21Z","abstract_excerpt":"Weight averaging has become a standard technique for enhancing model performance. However, methods such as Stochastic Weight Averaging (SWA) and Latest Weight Averaging (LAWA) often require manually designed procedures to sample from the training trajectory, and the results depend heavily on hyperparameter tuning. To minimize human effort, this paper proposes a simple yet efficient algorithm called Selective Weight Averaging (SeWA), which adaptively selects checkpoints during the final stages of training for averaging. Based on SeWA, we show that only a few points are needed to achieve better "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.10119","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.10119/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.10119","created_at":"2026-07-05T10:14:28.500300+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.10119v1","created_at":"2026-07-05T10:14:28.500300+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.10119","created_at":"2026-07-05T10:14:28.500300+00:00"},{"alias_kind":"pith_short_12","alias_value":"L4LPLZZC5ZSY","created_at":"2026-07-05T10:14:28.500300+00:00"},{"alias_kind":"pith_short_16","alias_value":"L4LPLZZC5ZSY2SBK","created_at":"2026-07-05T10:14:28.500300+00:00"},{"alias_kind":"pith_short_8","alias_value":"L4LPLZZC","created_at":"2026-07-05T10:14:28.500300+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24244","citing_title":"Model Merging Scaling Laws in Large Language Models","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK","json":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK.json","graph_json":"https://pith.science/api/pith-number/L4LPLZZC5ZSY2SBKIGBEO5FUDK/graph.json","events_json":"https://pith.science/api/pith-number/L4LPLZZC5ZSY2SBKIGBEO5FUDK/events.json","paper":"https://pith.science/paper/L4LPLZZC"},"agent_actions":{"view_html":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK","download_json":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK.json","view_paper":"https://pith.science/paper/L4LPLZZC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.10119&json=true","fetch_graph":"https://pith.science/api/pith-number/L4LPLZZC5ZSY2SBKIGBEO5FUDK/graph.json","fetch_events":"https://pith.science/api/pith-number/L4LPLZZC5ZSY2SBKIGBEO5FUDK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK/action/storage_attestation","attest_author":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK/action/author_attestation","sign_citation":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK/action/citation_signature","submit_replication":"https://pith.science/pith/L4LPLZZC5ZSY2SBKIGBEO5FUDK/action/replication_record"}},"created_at":"2026-07-05T10:14:28.500300+00:00","updated_at":"2026-07-05T10:14:28.500300+00:00"}