{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LJPVDZY5GXNNSAXZWKRS4XB66M","short_pith_number":"pith:LJPVDZY5","schema_version":"1.0","canonical_sha256":"5a5f51e71d35dad902f9b2a32e5c3ef32dbea4051da57779aec27daf76ed87f6","source":{"kind":"arxiv","id":"2502.06042","version":2},"attestation_state":"computed","paper":{"title":"Scaling Laws for Forgetting during Finetuning with Pretraining Data Injection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Dan Busbridge, David Grangier, Eleonora Gualdoni, Louis Bethune, Marco Cuturi, Pierre Ablin","submitted_at":"2025-02-09T21:44:27Z","abstract_excerpt":"A widespread strategy to obtain a language model that performs well on a target domain is to finetune a pretrained model to perform unsupervised next-token prediction on data from that target domain. Finetuning presents two challenges: (i) if the amount of target data is limited, as in most practical applications, the model will quickly overfit, and (ii) the model will drift away from the original model, forgetting the pretraining data and the generic knowledge that comes with it. We aim to derive scaling laws that quantify these two phenomena for various target domains, amounts of available t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06042","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-09T21:44:27Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5428bf2b0b98b9a19eadc453b2e808807f2ea803fbaa7460547a82257f095812","abstract_canon_sha256":"50c96e441c5c4e7b50bfd37cbddb833fa30174883561c29add0b52ef221ae119"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:04.604191Z","signature_b64":"igN7oZ0ORm6wQrFEZKsdWaQlG1mKnEFWC3EL8CZDNwK9QKLj/ZAjrVxe9mD2PgWGjReWxzpQF2DRqdDom4YrAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a5f51e71d35dad902f9b2a32e5c3ef32dbea4051da57779aec27daf76ed87f6","last_reissued_at":"2026-07-05T11:10:04.603666Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:04.603666Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Laws for Forgetting during Finetuning with Pretraining Data Injection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Dan Busbridge, David Grangier, Eleonora Gualdoni, Louis Bethune, Marco Cuturi, Pierre Ablin","submitted_at":"2025-02-09T21:44:27Z","abstract_excerpt":"A widespread strategy to obtain a language model that performs well on a target domain is to finetune a pretrained model to perform unsupervised next-token prediction on data from that target domain. Finetuning presents two challenges: (i) if the amount of target data is limited, as in most practical applications, the model will quickly overfit, and (ii) the model will drift away from the original model, forgetting the pretraining data and the generic knowledge that comes with it. We aim to derive scaling laws that quantify these two phenomena for various target domains, amounts of available t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06042","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06042/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06042","created_at":"2026-07-05T11:10:04.603735+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06042v2","created_at":"2026-07-05T11:10:04.603735+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06042","created_at":"2026-07-05T11:10:04.603735+00:00"},{"alias_kind":"pith_short_12","alias_value":"LJPVDZY5GXNN","created_at":"2026-07-05T11:10:04.603735+00:00"},{"alias_kind":"pith_short_16","alias_value":"LJPVDZY5GXNNSAXZ","created_at":"2026-07-05T11:10:04.603735+00:00"},{"alias_kind":"pith_short_8","alias_value":"LJPVDZY5","created_at":"2026-07-05T11:10:04.603735+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03924","citing_title":"Knowledge Editing in Masked Diffusion Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26097","citing_title":"Forgetting in Language Models: Capacity, Optimization, and Self-Generated Replay","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12715","citing_title":"Scaling Laws for Mixture Pretraining Under Data Constraints","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12705","citing_title":"Early Data Exposure Improves Robustness to Subsequent Fine-Tuning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M","json":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M.json","graph_json":"https://pith.science/api/pith-number/LJPVDZY5GXNNSAXZWKRS4XB66M/graph.json","events_json":"https://pith.science/api/pith-number/LJPVDZY5GXNNSAXZWKRS4XB66M/events.json","paper":"https://pith.science/paper/LJPVDZY5"},"agent_actions":{"view_html":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M","download_json":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M.json","view_paper":"https://pith.science/paper/LJPVDZY5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06042&json=true","fetch_graph":"https://pith.science/api/pith-number/LJPVDZY5GXNNSAXZWKRS4XB66M/graph.json","fetch_events":"https://pith.science/api/pith-number/LJPVDZY5GXNNSAXZWKRS4XB66M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M/action/storage_attestation","attest_author":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M/action/author_attestation","sign_citation":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M/action/citation_signature","submit_replication":"https://pith.science/pith/LJPVDZY5GXNNSAXZWKRS4XB66M/action/replication_record"}},"created_at":"2026-07-05T11:10:04.603735+00:00","updated_at":"2026-07-05T11:10:04.603735+00:00"}