{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M4CVSHAWJVVMXEXJXFZIVE35CK","short_pith_number":"pith:M4CVSHAW","schema_version":"1.0","canonical_sha256":"6705591c164d6acb92e9b9728a937d12b695e665298d0301d0079aedde63a1b9","source":{"kind":"arxiv","id":"2506.01901","version":1},"attestation_state":"computed","paper":{"title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chenlu Ye, Hanning Zhang, Rui Pan, Tong Zhang, Xingyuan Pan, Yifan Hao","submitted_at":"2025-06-02T17:23:16Z","abstract_excerpt":"Supervised fine-tuning (SFT) on domain-specific data is the dominant approach for adapting foundation models to specialized tasks. However, it has been observed that SFT models tend to forget knowledge acquired during pretraining. In vision models, ensembling a pretrained model with its fine-tuned counterpart has been shown to mitigate this issue. In this work, we demonstrate that the same holds for language models, and, more strikingly, we observe an overadaptation phenomenon: the ensemble model not only retains general knowledge from the foundation model but also outperforms the fine-tuned m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.01901","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-02T17:23:16Z","cross_cats_sorted":[],"title_canon_sha256":"c484058ddac077584fccdc790ae552097b4e31853c5bb36ea08f7b10efe08ed4","abstract_canon_sha256":"b63a852f2b18b8237a1da5e8bd6f21855610a555fb3d9e585e0fe4ac3445e972"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:21.020989Z","signature_b64":"GFXUqxNSHNQBTldyH40+yymnMIERrcs6N4ZF0wCZIe9j2epMmWn852PZdrEHCdg2GXFIJ/QkiYyMi4Wy1XZ/Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6705591c164d6acb92e9b9728a937d12b695e665298d0301d0079aedde63a1b9","last_reissued_at":"2026-07-05T11:14:21.020563Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:21.020563Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chenlu Ye, Hanning Zhang, Rui Pan, Tong Zhang, Xingyuan Pan, Yifan Hao","submitted_at":"2025-06-02T17:23:16Z","abstract_excerpt":"Supervised fine-tuning (SFT) on domain-specific data is the dominant approach for adapting foundation models to specialized tasks. However, it has been observed that SFT models tend to forget knowledge acquired during pretraining. In vision models, ensembling a pretrained model with its fine-tuned counterpart has been shown to mitigate this issue. In this work, we demonstrate that the same holds for language models, and, more strikingly, we observe an overadaptation phenomenon: the ensemble model not only retains general knowledge from the foundation model but also outperforms the fine-tuned m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.01901","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.01901/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.01901","created_at":"2026-07-05T11:14:21.020615+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.01901v1","created_at":"2026-07-05T11:14:21.020615+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.01901","created_at":"2026-07-05T11:14:21.020615+00:00"},{"alias_kind":"pith_short_12","alias_value":"M4CVSHAWJVVM","created_at":"2026-07-05T11:14:21.020615+00:00"},{"alias_kind":"pith_short_16","alias_value":"M4CVSHAWJVVMXEXJ","created_at":"2026-07-05T11:14:21.020615+00:00"},{"alias_kind":"pith_short_8","alias_value":"M4CVSHAW","created_at":"2026-07-05T11:14:21.020615+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21890","citing_title":"Scaling Performance and Low-Resource Annotation with Many-Shot In-Context Learning for Named Entity Recognition","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22126","citing_title":"AesFormer: Transform Everyday Photos into Beautiful Memories","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01954","citing_title":"Moira: Language-driven Hierarchical Reinforcement Learning for Pair Trading","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK","json":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK.json","graph_json":"https://pith.science/api/pith-number/M4CVSHAWJVVMXEXJXFZIVE35CK/graph.json","events_json":"https://pith.science/api/pith-number/M4CVSHAWJVVMXEXJXFZIVE35CK/events.json","paper":"https://pith.science/paper/M4CVSHAW"},"agent_actions":{"view_html":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK","download_json":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK.json","view_paper":"https://pith.science/paper/M4CVSHAW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.01901&json=true","fetch_graph":"https://pith.science/api/pith-number/M4CVSHAWJVVMXEXJXFZIVE35CK/graph.json","fetch_events":"https://pith.science/api/pith-number/M4CVSHAWJVVMXEXJXFZIVE35CK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK/action/storage_attestation","attest_author":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK/action/author_attestation","sign_citation":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK/action/citation_signature","submit_replication":"https://pith.science/pith/M4CVSHAWJVVMXEXJXFZIVE35CK/action/replication_record"}},"created_at":"2026-07-05T11:14:21.020615+00:00","updated_at":"2026-07-05T11:14:21.020615+00:00"}