{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:QCRY3IBVQ2I322UABORPZ2YH7S","short_pith_number":"pith:QCRY3IBV","schema_version":"1.0","canonical_sha256":"80a38da0358691bd6a800ba2fceb07fc969f08c77ddce73fa45259021e916035","source":{"kind":"arxiv","id":"2012.13255","version":1},"attestation_state":"computed","paper":{"title":"Intrinsic Dimensionality Explains the Effectiveness of Language Model Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Armen Aghajanyan, Luke Zettlemoyer, Sonal Gupta","submitted_at":"2020-12-22T07:42:30Z","abstract_excerpt":"Although pretrained language models can be fine-tuned to produce state-of-the-art results for a very wide range of language understanding tasks, the dynamics of this process are not well understood, especially in the low data regime. Why can we use relatively vanilla gradient descent algorithms (e.g., without strong regularization) to tune a model with hundreds of millions of parameters on datasets with only hundreds or thousands of labeled examples? In this paper, we argue that analyzing fine-tuning through the lens of intrinsic dimension provides us with empirical and theoretical intuitions "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.13255","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2020-12-22T07:42:30Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2ec75423c3abf5bd3bc2eba4afe2ae179c11eefe3d536d7797da052370673048","abstract_canon_sha256":"a7481c4fe7e01b54e067b853b211c62f80bebcf8b6cc52407f2e21e09d35cd4e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:01:56.744597Z","signature_b64":"Hezv67L3zLsaofYVBDRNCFg4Emb7dCsf1bCSGNc2Zm/vxzTCMG4VuDhI07V1ywFsol1zOgNdy4ic6zFS6gOYCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80a38da0358691bd6a800ba2fceb07fc969f08c77ddce73fa45259021e916035","last_reissued_at":"2026-07-05T02:01:56.744074Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:01:56.744074Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Intrinsic Dimensionality Explains the Effectiveness of Language Model Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Armen Aghajanyan, Luke Zettlemoyer, Sonal Gupta","submitted_at":"2020-12-22T07:42:30Z","abstract_excerpt":"Although pretrained language models can be fine-tuned to produce state-of-the-art results for a very wide range of language understanding tasks, the dynamics of this process are not well understood, especially in the low data regime. Why can we use relatively vanilla gradient descent algorithms (e.g., without strong regularization) to tune a model with hundreds of millions of parameters on datasets with only hundreds or thousands of labeled examples? In this paper, we argue that analyzing fine-tuning through the lens of intrinsic dimension provides us with empirical and theoretical intuitions "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.13255","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.13255/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.13255","created_at":"2026-07-05T02:01:56.744160+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.13255v1","created_at":"2026-07-05T02:01:56.744160+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.13255","created_at":"2026-07-05T02:01:56.744160+00:00"},{"alias_kind":"pith_short_12","alias_value":"QCRY3IBVQ2I3","created_at":"2026-07-05T02:01:56.744160+00:00"},{"alias_kind":"pith_short_16","alias_value":"QCRY3IBVQ2I322UA","created_at":"2026-07-05T02:01:56.744160+00:00"},{"alias_kind":"pith_short_8","alias_value":"QCRY3IBV","created_at":"2026-07-05T02:01:56.744160+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00162","citing_title":"FRAME: Learning the Adaptation Domain with a Mixture of Fractional-Fourier Experts","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06920","citing_title":"The Fine-Tuning Trap: Evaluating Negative Transfer and the Role of PEFT in Sub-1B Mathematical Reasoning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05465","citing_title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2505.01307","citing_title":"Document Retrieval Augmented Fine-Tuning (DRAFT) for safety-critical software assessments","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2505.03205","citing_title":"Transformers for Learning on Noisy and Task-Level Manifolds: Approximation and Generalization Insights","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21035","citing_title":"Little by Little: Continual Learning via Incremental Mixture of Rank-1 Associative Memory Experts","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20693","citing_title":"Interpretable Discriminative Text Representations via Agreement and Label Disentanglement","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15916","citing_title":"LoCO: Low-rank Compositional Rotation Fine-tuning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18993","citing_title":"CR-Net: Scaling Parameter-Efficient Training with Cross-Layer Low-Rank Structure","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18629","citing_title":"HyperAdapt: Simple High-Rank Adaptation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06179","citing_title":"ARIA: Adaptive Retrieval Intelligence Assistant -- A Multimodal RAG Framework for Domain-Specific Engineering Education","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13421","citing_title":"Combining pre-trained models via localized model averaging","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14608","citing_title":"Parameter-Efficient Fine-Tuning for Large Models: A Comprehensive Survey","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05687","citing_title":"DataDignity: Training Data Attribution for Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2101.00190","citing_title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00658","citing_title":"UniVidX: A Unified Multimodal Framework for Versatile Video Generation via Diffusion Priors","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07407","citing_title":"Emergent Symbolic Structure in Health Foundation Models: Extraction, Alignment, and Cross-Modal Transfer","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04440","citing_title":"Training Transformers in Cosine Coefficient Space","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18124","citing_title":"TLoRA: Task-aware Low Rank Adaptation of Large Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19118","citing_title":"DP-FlogTinyLLM: Differentially private federated log anomaly detection using Tiny LLMs","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2106.09685","citing_title":"LoRA: Low-Rank Adaptation of Large Language Models","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S","json":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S.json","graph_json":"https://pith.science/api/pith-number/QCRY3IBVQ2I322UABORPZ2YH7S/graph.json","events_json":"https://pith.science/api/pith-number/QCRY3IBVQ2I322UABORPZ2YH7S/events.json","paper":"https://pith.science/paper/QCRY3IBV"},"agent_actions":{"view_html":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S","download_json":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S.json","view_paper":"https://pith.science/paper/QCRY3IBV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.13255&json=true","fetch_graph":"https://pith.science/api/pith-number/QCRY3IBVQ2I322UABORPZ2YH7S/graph.json","fetch_events":"https://pith.science/api/pith-number/QCRY3IBVQ2I322UABORPZ2YH7S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S/action/storage_attestation","attest_author":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S/action/author_attestation","sign_citation":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S/action/citation_signature","submit_replication":"https://pith.science/pith/QCRY3IBVQ2I322UABORPZ2YH7S/action/replication_record"}},"created_at":"2026-07-05T02:01:56.744160+00:00","updated_at":"2026-07-05T02:01:56.744160+00:00"}