{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:EYCV5NY64RXW6X7L5YGLJJMDC3","short_pith_number":"pith:EYCV5NY6","schema_version":"1.0","canonical_sha256":"26055eb71ee46f6f5febee0cb4a58316f407f530f42fd944f95103099728cd2b","source":{"kind":"arxiv","id":"2110.04366","version":3},"attestation_state":"computed","paper":{"title":"Towards a Unified View of Parameter-Efficient Transfer Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Graham Neubig, Junxian He, Taylor Berg-Kirkpatrick, Xuezhe Ma","submitted_at":"2021-10-08T20:22:26Z","abstract_excerpt":"Fine-tuning large pre-trained language models on downstream tasks has become the de-facto learning paradigm in NLP. However, conventional approaches fine-tune all the parameters of the pre-trained model, which becomes prohibitive as the model size and the number of tasks grow. Recent work has proposed a variety of parameter-efficient transfer learning methods that only fine-tune a small number of (extra) parameters to attain strong performance. While effective, the critical ingredients for success and the connections among the various methods are poorly understood. In this paper, we break down"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.04366","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-10-08T20:22:26Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"85ed685b1bed952763a6c8880ff93e7e21473016ac6f63992fe614d83ba38104","abstract_canon_sha256":"2a28b0a046af80af58f98e3aff4aab57c8faaf7d446cfb9c06f15ce4c6c5df6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:53:38.243915Z","signature_b64":"2Ioj3v9h4+S0yCj25od4xlL/ErU5x5AXi6+Ag+kgDvSYFV2eypWwIVf/7zq6PEXMW+2kEsgZG1CBYDsrZnphBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26055eb71ee46f6f5febee0cb4a58316f407f530f42fd944f95103099728cd2b","last_reissued_at":"2026-07-05T03:53:38.243372Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:53:38.243372Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards a Unified View of Parameter-Efficient Transfer Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Graham Neubig, Junxian He, Taylor Berg-Kirkpatrick, Xuezhe Ma","submitted_at":"2021-10-08T20:22:26Z","abstract_excerpt":"Fine-tuning large pre-trained language models on downstream tasks has become the de-facto learning paradigm in NLP. However, conventional approaches fine-tune all the parameters of the pre-trained model, which becomes prohibitive as the model size and the number of tasks grow. Recent work has proposed a variety of parameter-efficient transfer learning methods that only fine-tune a small number of (extra) parameters to attain strong performance. While effective, the critical ingredients for success and the connections among the various methods are poorly understood. In this paper, we break down"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.04366","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.04366/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.04366","created_at":"2026-07-05T03:53:38.243433+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.04366v3","created_at":"2026-07-05T03:53:38.243433+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.04366","created_at":"2026-07-05T03:53:38.243433+00:00"},{"alias_kind":"pith_short_12","alias_value":"EYCV5NY64RXW","created_at":"2026-07-05T03:53:38.243433+00:00"},{"alias_kind":"pith_short_16","alias_value":"EYCV5NY64RXW6X7L","created_at":"2026-07-05T03:53:38.243433+00:00"},{"alias_kind":"pith_short_8","alias_value":"EYCV5NY6","created_at":"2026-07-05T03:53:38.243433+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24161","citing_title":"Dual-Branch Cross-Projection Debiasing through Diffusion-based Disentanglement","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20553","citing_title":"From Efficiency to Leakage -- Privacy Backdoor in Federated Language Model Fine-Tuning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01858","citing_title":"Polaris: Scaling Up Instruction-Guided Image Generation Towards Millions of Personalized Style Needs","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31813","citing_title":"Geometry-Preserving Orthonormal Initialization for Low-Rank Adaptation in RLVR","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05465","citing_title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15214","citing_title":"Histogram-based Parameter-efficient Tuning for Passive and Active Sonar Classification","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04227","citing_title":"Continual Learning for VLMs: A Survey and Taxonomy Beyond Forgetting","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15615","citing_title":"Neutral-Reference Prompting for Vision-Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2510.24561","citing_title":"LoRA-DA: Data-Aware Initialization for Low-Rank Adaptation via Asymptotic Analysis","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2404.02258","citing_title":"Mixture-of-Depths: Dynamically allocating compute in transformer-based language models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.22911","citing_title":"CeRA: Breaking the Linear Ceiling of Low-Rank Adaptation with Non-linearity Retained at Inference","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2305.18323","citing_title":"ReWOO: Decoupling Reasoning from Observations for Efficient Augmented Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14055","citing_title":"PEML: Parameter-efficient Multi-Task Learning with Optimized Continuous Prompts","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14608","citing_title":"Parameter-Efficient Fine-Tuning for Large Models: A Comprehensive Survey","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05910","citing_title":"Plug-and-play Class-aware Knowledge Injection for Prompt Learning with Visual-Language Model","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19015","citing_title":"FedProxy: Federated Fine-Tuning of LLMs via Proxy SLMs and Heterogeneity-Aware Fusion","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12110","citing_title":"SOLARIS: Speculative Offloading of Latent-bAsed Representation for Inference Scaling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12502","citing_title":"SEATrack: Simple, Efficient, and Adaptive Multimodal Tracker","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09034","citing_title":"The nextAI Solution to the NeurIPS 2023 LLM Efficiency Challenge","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07302","citing_title":"Pretraining Induces a Reusable Spectral Basis for Downstream Task Adaptation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06440","citing_title":"Visual prompting reimagined: The power of the Activation Prompts","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02713","citing_title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18124","citing_title":"TLoRA: Task-aware Low Rank Adaptation of Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04447","citing_title":"Deep Reprogramming Distillation for Medical Foundation Models","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3","json":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3.json","graph_json":"https://pith.science/api/pith-number/EYCV5NY64RXW6X7L5YGLJJMDC3/graph.json","events_json":"https://pith.science/api/pith-number/EYCV5NY64RXW6X7L5YGLJJMDC3/events.json","paper":"https://pith.science/paper/EYCV5NY6"},"agent_actions":{"view_html":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3","download_json":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3.json","view_paper":"https://pith.science/paper/EYCV5NY6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.04366&json=true","fetch_graph":"https://pith.science/api/pith-number/EYCV5NY64RXW6X7L5YGLJJMDC3/graph.json","fetch_events":"https://pith.science/api/pith-number/EYCV5NY64RXW6X7L5YGLJJMDC3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3/action/storage_attestation","attest_author":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3/action/author_attestation","sign_citation":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3/action/citation_signature","submit_replication":"https://pith.science/pith/EYCV5NY64RXW6X7L5YGLJJMDC3/action/replication_record"}},"created_at":"2026-07-05T03:53:38.243433+00:00","updated_at":"2026-07-05T03:53:38.243433+00:00"}