{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7IUGSMZEXTPYH3CL4UU4KOJ5GD","short_pith_number":"pith:7IUGSMZE","schema_version":"1.0","canonical_sha256":"fa28693324bcdf83ec4be529c5393d30da2098091c80c97f821f995642126f76","source":{"kind":"arxiv","id":"2302.06354","version":3},"attestation_state":"computed","paper":{"title":"Less is More: Selective Layer Finetuning with SubTuning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrey Gurevich, Eran Malach, Gal Kaplun, Mazor David, Shai Shalev-Shwartz, Tal Swisa","submitted_at":"2023-02-13T13:38:46Z","abstract_excerpt":"Finetuning a pretrained model has become a standard approach for training neural networks on novel tasks, resulting in fast convergence and improved performance. In this work, we study an alternative finetuning method, where instead of finetuning all the weights of the network, we only train a carefully chosen subset of layers, keeping the rest of the weights frozen at their initial (pretrained) values. We demonstrate that \\emph{subset finetuning} (or SubTuning) often achieves accuracy comparable to full finetuning of the model, and even surpasses the performance of full finetuning when traini"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.06354","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-13T13:38:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"22aeecaf1fe2f7c2952b23622ba8a08a6da669ba321417a2e89dfbe4e8be498c","abstract_canon_sha256":"77937fcc4a2983d05b2abf62b1e9dc7c3d881341dd6780d1485ffe1a687cb4e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:26:45.730847Z","signature_b64":"fKxo+VCnskagb/y7ysRQ77V+WoLPFuCv2S9aP8xtO2dSSA9jIGf10n+6BOyup1I2TDJyQogOO027fCBezUECAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa28693324bcdf83ec4be529c5393d30da2098091c80c97f821f995642126f76","last_reissued_at":"2026-07-05T06:26:45.730356Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:26:45.730356Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Less is More: Selective Layer Finetuning with SubTuning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andrey Gurevich, Eran Malach, Gal Kaplun, Mazor David, Shai Shalev-Shwartz, Tal Swisa","submitted_at":"2023-02-13T13:38:46Z","abstract_excerpt":"Finetuning a pretrained model has become a standard approach for training neural networks on novel tasks, resulting in fast convergence and improved performance. In this work, we study an alternative finetuning method, where instead of finetuning all the weights of the network, we only train a carefully chosen subset of layers, keeping the rest of the weights frozen at their initial (pretrained) values. We demonstrate that \\emph{subset finetuning} (or SubTuning) often achieves accuracy comparable to full finetuning of the model, and even surpasses the performance of full finetuning when traini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.06354","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.06354/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.06354","created_at":"2026-07-05T06:26:45.730411+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.06354v3","created_at":"2026-07-05T06:26:45.730411+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.06354","created_at":"2026-07-05T06:26:45.730411+00:00"},{"alias_kind":"pith_short_12","alias_value":"7IUGSMZEXTPY","created_at":"2026-07-05T06:26:45.730411+00:00"},{"alias_kind":"pith_short_16","alias_value":"7IUGSMZEXTPYH3CL","created_at":"2026-07-05T06:26:45.730411+00:00"},{"alias_kind":"pith_short_8","alias_value":"7IUGSMZE","created_at":"2026-07-05T06:26:45.730411+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31397","citing_title":"Mixture-of-Control: State-Aware Fine-Tuning for Transformer-based Models","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19262","citing_title":"Backdooring Masked Diffusion Language Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19262","citing_title":"Backdooring Masked Diffusion Language Models","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD","json":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD.json","graph_json":"https://pith.science/api/pith-number/7IUGSMZEXTPYH3CL4UU4KOJ5GD/graph.json","events_json":"https://pith.science/api/pith-number/7IUGSMZEXTPYH3CL4UU4KOJ5GD/events.json","paper":"https://pith.science/paper/7IUGSMZE"},"agent_actions":{"view_html":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD","download_json":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD.json","view_paper":"https://pith.science/paper/7IUGSMZE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.06354&json=true","fetch_graph":"https://pith.science/api/pith-number/7IUGSMZEXTPYH3CL4UU4KOJ5GD/graph.json","fetch_events":"https://pith.science/api/pith-number/7IUGSMZEXTPYH3CL4UU4KOJ5GD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD/action/storage_attestation","attest_author":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD/action/author_attestation","sign_citation":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD/action/citation_signature","submit_replication":"https://pith.science/pith/7IUGSMZEXTPYH3CL4UU4KOJ5GD/action/replication_record"}},"created_at":"2026-07-05T06:26:45.730411+00:00","updated_at":"2026-07-05T06:26:45.730411+00:00"}