{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PQLY2IQVHZFAMNRXLZF23U46OU","short_pith_number":"pith:PQLY2IQV","schema_version":"1.0","canonical_sha256":"7c178d22153e4a0636375e4badd39e75064124be61aeb858dbbafacd0756ed93","source":{"kind":"arxiv","id":"2402.08096","version":3},"attestation_state":"computed","paper":{"title":"An Efficient Rehearsal Scheme for Catastrophic Forgetting Mitigation during Multi-stage Fine-tuning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Bai, Ankur Taly, Chih-Kuan Yeh, Cho-Jui Hsieh","submitted_at":"2024-02-12T22:32:12Z","abstract_excerpt":"Incrementally fine-tuning foundational models on new tasks or domains is now the de facto approach in NLP. A known pitfall of this approach is the \\emph{catastrophic forgetting} of prior knowledge that happens during fine-tuning. A common approach to alleviate such forgetting is to rehearse samples from prior tasks during fine-tuning. Several existing works assume a fixed memory buffer to store prior task examples, while relying on inferences (forward passes) with the model at hand for choosing examples for rehearsal from the buffer. However, given the increasing computational cost of model in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08096","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-12T22:32:12Z","cross_cats_sorted":[],"title_canon_sha256":"ac0301197141da7c806a87c588e977e2acc94a4c1222ab87fbacc94a8ccd605d","abstract_canon_sha256":"38caef2380707de2eedac10e749235ed22677a892fb4b24c82c9cd6ced928224"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:46.941134Z","signature_b64":"mL5v9NCKaLF2wnb06p4jC/uTG50TOFL58f2AzoTV6KaKAk95dPsVC7JnynmM3GxAbyOOSwxtoHun562ihISUBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c178d22153e4a0636375e4badd39e75064124be61aeb858dbbafacd0756ed93","last_reissued_at":"2026-07-05T10:12:46.939126Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:46.939126Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Efficient Rehearsal Scheme for Catastrophic Forgetting Mitigation during Multi-stage Fine-tuning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Bai, Ankur Taly, Chih-Kuan Yeh, Cho-Jui Hsieh","submitted_at":"2024-02-12T22:32:12Z","abstract_excerpt":"Incrementally fine-tuning foundational models on new tasks or domains is now the de facto approach in NLP. A known pitfall of this approach is the \\emph{catastrophic forgetting} of prior knowledge that happens during fine-tuning. A common approach to alleviate such forgetting is to rehearse samples from prior tasks during fine-tuning. Several existing works assume a fixed memory buffer to store prior task examples, while relying on inferences (forward passes) with the model at hand for choosing examples for rehearsal from the buffer. However, given the increasing computational cost of model in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08096","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08096/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08096","created_at":"2026-07-05T10:12:46.939192+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08096v3","created_at":"2026-07-05T10:12:46.939192+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08096","created_at":"2026-07-05T10:12:46.939192+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQLY2IQVHZFA","created_at":"2026-07-05T10:12:46.939192+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQLY2IQVHZFAMNRX","created_at":"2026-07-05T10:12:46.939192+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQLY2IQV","created_at":"2026-07-05T10:12:46.939192+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02079","citing_title":"FACT: A Simple and Efficient Framework for Active Finetuning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03938","citing_title":"FOREVER: Forgetting Curve-Inspired Memory Replay for Language Model Continual Learning","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU","json":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU.json","graph_json":"https://pith.science/api/pith-number/PQLY2IQVHZFAMNRXLZF23U46OU/graph.json","events_json":"https://pith.science/api/pith-number/PQLY2IQVHZFAMNRXLZF23U46OU/events.json","paper":"https://pith.science/paper/PQLY2IQV"},"agent_actions":{"view_html":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU","download_json":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU.json","view_paper":"https://pith.science/paper/PQLY2IQV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08096&json=true","fetch_graph":"https://pith.science/api/pith-number/PQLY2IQVHZFAMNRXLZF23U46OU/graph.json","fetch_events":"https://pith.science/api/pith-number/PQLY2IQVHZFAMNRXLZF23U46OU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU/action/storage_attestation","attest_author":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU/action/author_attestation","sign_citation":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU/action/citation_signature","submit_replication":"https://pith.science/pith/PQLY2IQVHZFAMNRXLZF23U46OU/action/replication_record"}},"created_at":"2026-07-05T10:12:46.939192+00:00","updated_at":"2026-07-05T10:12:46.939192+00:00"}