{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5QFDM4HMAXE7R43OVHUFAY67OY","short_pith_number":"pith:5QFDM4HM","schema_version":"1.0","canonical_sha256":"ec0a3670ec05c9f8f36ea9e85063df7605a8f22f99d7ab28eee8a69e15a1f8f0","source":{"kind":"arxiv","id":"2405.09673","version":2},"attestation_state":"computed","paper":{"title":"LoRA Learns Less and Forgets Less","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Cody Blakeney, Connor Jennings, Dan Biderman, Daniel King, Jacob Portes, John P. Cunningham, Jonathan Frankle, Jose Javier Gonzalez Ortiz, Mansheej Paul, Philip Greengard, Sam Havens, Vitaliy Chiley","submitted_at":"2024-05-15T19:27:45Z","abstract_excerpt":"Low-Rank Adaptation (LoRA) is a widely-used parameter-efficient finetuning method for large language models. LoRA saves memory by training only low rank perturbations to selected weight matrices. In this work, we compare the performance of LoRA and full finetuning on two target domains, programming and mathematics. We consider both the instruction finetuning (approximately 100K prompt-response pairs) and continued pretraining (20B unstructured tokens) data regimes. Our results show that, in the standard low-rank settings, LoRA substantially underperforms full finetuning. Nevertheless, LoRA bet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.09673","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-15T19:27:45Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d1388b4309332395b7a91fcb8d1b7ea0bb96cb2d9a7658245f8df82a940ea76b","abstract_canon_sha256":"c5655e5317910113229ef061691dd1f99a0b3a9f0381965916ae9ebdfb7d7aea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:09:59.531550Z","signature_b64":"w8fPjcLaaQpOrQOoeIZpZ/DrjGPqI3zwZIDPzRZVVOWlV8j42Ipsc+TlGcHodhpnp3B1bhgaBk3PggUlhZRMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec0a3670ec05c9f8f36ea9e85063df7605a8f22f99d7ab28eee8a69e15a1f8f0","last_reissued_at":"2026-07-05T09:09:59.531057Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:09:59.531057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LoRA Learns Less and Forgets Less","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Cody Blakeney, Connor Jennings, Dan Biderman, Daniel King, Jacob Portes, John P. Cunningham, Jonathan Frankle, Jose Javier Gonzalez Ortiz, Mansheej Paul, Philip Greengard, Sam Havens, Vitaliy Chiley","submitted_at":"2024-05-15T19:27:45Z","abstract_excerpt":"Low-Rank Adaptation (LoRA) is a widely-used parameter-efficient finetuning method for large language models. LoRA saves memory by training only low rank perturbations to selected weight matrices. In this work, we compare the performance of LoRA and full finetuning on two target domains, programming and mathematics. We consider both the instruction finetuning (approximately 100K prompt-response pairs) and continued pretraining (20B unstructured tokens) data regimes. Our results show that, in the standard low-rank settings, LoRA substantially underperforms full finetuning. Nevertheless, LoRA bet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.09673","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.09673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.09673","created_at":"2026-07-05T09:09:59.531117+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.09673v2","created_at":"2026-07-05T09:09:59.531117+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.09673","created_at":"2026-07-05T09:09:59.531117+00:00"},{"alias_kind":"pith_short_12","alias_value":"5QFDM4HMAXE7","created_at":"2026-07-05T09:09:59.531117+00:00"},{"alias_kind":"pith_short_16","alias_value":"5QFDM4HMAXE7R43O","created_at":"2026-07-05T09:09:59.531117+00:00"},{"alias_kind":"pith_short_8","alias_value":"5QFDM4HM","created_at":"2026-07-05T09:09:59.531117+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25342","citing_title":"Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05559","citing_title":"CLaaS: Continual learning as a service for sample efficient online learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02437","citing_title":"On the Scaling of PEFT: Towards Million Personal Models of Trillion Parameters","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29498","citing_title":"Mask the Target: A Plug-and-Play Regularizer Against LoRA Forgetting","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21035","citing_title":"Little by Little: Continual Learning via Incremental Mixture of Rank-1 Associative Memory Experts","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04227","citing_title":"Continual Learning for VLMs: A Survey and Taxonomy Beyond Forgetting","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07111","citing_title":"Beyond LoRA vs. Full Fine-Tuning: Gradient-Guided Optimizer Routing for LLM Adaptation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20005","citing_title":"Fine-Tuning Without Forgetting via Loss-Adaptive Learning Rates","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00029","citing_title":"LoRA-Mixer: Coordinate Modular LoRA Experts Through Serial Attention Routing","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18993","citing_title":"CR-Net: Scaling Parameter-Efficient Training with Cross-Layer Low-Rank Structure","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21788","citing_title":"InstructMoLE: Instruction-Guided Mixture of Low-rank Experts for Multi-Conditional Image Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12705","citing_title":"Early Data Exposure Improves Robustness to Subsequent Fine-Tuning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12492","citing_title":"Pion: A Spectrum-Preserving Optimizer via Orthogonal Equivalence Transformation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09015","citing_title":"LLiMba: Sardinian on a Single GPU -- Adapting a 3B Language Model to a Vanishing Romance Language","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06654","citing_title":"Optimizer-Model Consistency: Full Finetuning with the Same Optimizer as Pretraining Forgets Less","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20720","citing_title":"COMPASS: COntinual Multilingual PEFT with Adaptive Semantic Sampling","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12426","citing_title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07111","citing_title":"Beyond LoRA vs. Full Fine-Tuning: Gradient-Guided Optimizer Routing for LLM Adaptation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18124","citing_title":"TLoRA: Task-aware Low Rank Adaptation of Large Language Models","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY","json":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY.json","graph_json":"https://pith.science/api/pith-number/5QFDM4HMAXE7R43OVHUFAY67OY/graph.json","events_json":"https://pith.science/api/pith-number/5QFDM4HMAXE7R43OVHUFAY67OY/events.json","paper":"https://pith.science/paper/5QFDM4HM"},"agent_actions":{"view_html":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY","download_json":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY.json","view_paper":"https://pith.science/paper/5QFDM4HM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.09673&json=true","fetch_graph":"https://pith.science/api/pith-number/5QFDM4HMAXE7R43OVHUFAY67OY/graph.json","fetch_events":"https://pith.science/api/pith-number/5QFDM4HMAXE7R43OVHUFAY67OY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY/action/storage_attestation","attest_author":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY/action/author_attestation","sign_citation":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY/action/citation_signature","submit_replication":"https://pith.science/pith/5QFDM4HMAXE7R43OVHUFAY67OY/action/replication_record"}},"created_at":"2026-07-05T09:09:59.531117+00:00","updated_at":"2026-07-05T09:09:59.531117+00:00"}