{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:D6AI3WW4GDF5M2Q7EKAARUG7EO","short_pith_number":"pith:D6AI3WW4","schema_version":"1.0","canonical_sha256":"1f808ddadc30cbd66a1f228008d0df23bd8073ce3764f8ccf93df61e4e52787a","source":{"kind":"arxiv","id":"2502.08130","version":2},"attestation_state":"computed","paper":{"title":"Selective Self-to-Supervised Fine-Tuning for Generalization in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Asaf Yehudai, Dinesh Khandelwal, Dinesh Raghu, Sachindra Joshi, Sonam Gupta, Yatin Nandwani","submitted_at":"2025-02-12T05:24:21Z","abstract_excerpt":"Fine-tuning Large Language Models (LLMs) on specific datasets is a common practice to improve performance on target tasks. However, this performance gain often leads to overfitting, where the model becomes too specialized in either the task or the characteristics of the training data, resulting in a loss of generalization. This paper introduces Selective Self-to-Supervised Fine-Tuning (S3FT), a fine-tuning approach that achieves better performance than the standard supervised fine-tuning (SFT) while improving generalization. S3FT leverages the existence of multiple valid responses to a query. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08130","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-12T05:24:21Z","cross_cats_sorted":[],"title_canon_sha256":"7b5859dc1e21c13e9091cf5d8d37e6f58e61829564e83f05add467bcb81eb92e","abstract_canon_sha256":"e5d42e4239679e5c49b2f318e6a47953e4213d75047c3f598f3fd86a201416af"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:16.656889Z","signature_b64":"dwTlk8JHZ2L0OEkz8XrRNjCsJ4V3yk/Ob7v0sa9xMZ6S8MzzUPSo6m7+iOxWAorsr2ykVQFS8OrEdYs5FoqYAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f808ddadc30cbd66a1f228008d0df23bd8073ce3764f8ccf93df61e4e52787a","last_reissued_at":"2026-07-05T10:17:16.656399Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:16.656399Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Selective Self-to-Supervised Fine-Tuning for Generalization in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Asaf Yehudai, Dinesh Khandelwal, Dinesh Raghu, Sachindra Joshi, Sonam Gupta, Yatin Nandwani","submitted_at":"2025-02-12T05:24:21Z","abstract_excerpt":"Fine-tuning Large Language Models (LLMs) on specific datasets is a common practice to improve performance on target tasks. However, this performance gain often leads to overfitting, where the model becomes too specialized in either the task or the characteristics of the training data, resulting in a loss of generalization. This paper introduces Selective Self-to-Supervised Fine-Tuning (S3FT), a fine-tuning approach that achieves better performance than the standard supervised fine-tuning (SFT) while improving generalization. S3FT leverages the existence of multiple valid responses to a query. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08130","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08130/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08130","created_at":"2026-07-05T10:17:16.656461+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08130v2","created_at":"2026-07-05T10:17:16.656461+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08130","created_at":"2026-07-05T10:17:16.656461+00:00"},{"alias_kind":"pith_short_12","alias_value":"D6AI3WW4GDF5","created_at":"2026-07-05T10:17:16.656461+00:00"},{"alias_kind":"pith_short_16","alias_value":"D6AI3WW4GDF5M2Q7","created_at":"2026-07-05T10:17:16.656461+00:00"},{"alias_kind":"pith_short_8","alias_value":"D6AI3WW4","created_at":"2026-07-05T10:17:16.656461+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11922","citing_title":"StepCodeReasoner: Aligning Code Reasoning with Stepwise Execution Traces via Reinforcement Learning","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO","json":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO.json","graph_json":"https://pith.science/api/pith-number/D6AI3WW4GDF5M2Q7EKAARUG7EO/graph.json","events_json":"https://pith.science/api/pith-number/D6AI3WW4GDF5M2Q7EKAARUG7EO/events.json","paper":"https://pith.science/paper/D6AI3WW4"},"agent_actions":{"view_html":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO","download_json":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO.json","view_paper":"https://pith.science/paper/D6AI3WW4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08130&json=true","fetch_graph":"https://pith.science/api/pith-number/D6AI3WW4GDF5M2Q7EKAARUG7EO/graph.json","fetch_events":"https://pith.science/api/pith-number/D6AI3WW4GDF5M2Q7EKAARUG7EO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO/action/storage_attestation","attest_author":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO/action/author_attestation","sign_citation":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO/action/citation_signature","submit_replication":"https://pith.science/pith/D6AI3WW4GDF5M2Q7EKAARUG7EO/action/replication_record"}},"created_at":"2026-07-05T10:17:16.656461+00:00","updated_at":"2026-07-05T10:17:16.656461+00:00"}