{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RT3L3OB6LGS7KDYQEP5GVVOH7Q","short_pith_number":"pith:RT3L3OB6","schema_version":"1.0","canonical_sha256":"8cf6bdb83e59a5f50f1023fa6ad5c7fc250794a103a210ca205d94b1dc53a9ab","source":{"kind":"arxiv","id":"2312.11420","version":1},"attestation_state":"computed","paper":{"title":"Tuning LayerNorm in Attention: Towards Efficient Multi-Modal LLM Finetuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Bingchen Zhao, Chen Wei, Cihang Xie, Haoqin Tu, Jieru Mei","submitted_at":"2023-12-18T18:21:43Z","abstract_excerpt":"This paper introduces an efficient strategy to transform Large Language Models (LLMs) into Multi-Modal Large Language Models (MLLMs). By conceptualizing this transformation as a domain adaptation process, i.e., transitioning from text understanding to embracing multiple modalities, we intriguingly note that, within each attention block, tuning LayerNorm suffices to yield strong performance. Moreover, when benchmarked against other tuning approaches like full parameter finetuning or LoRA, its benefits on efficiency are substantial. For example, when compared to LoRA on a 13B model scale, perfor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.11420","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-18T18:21:43Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"c4b3da3c6ef5902dca702b73db41a67a810158161669d6f8ffd16bf92b6e29f6","abstract_canon_sha256":"30983de377670ee4d06e96fa88b0326e6c5131621c2133f91177603ca9c30285"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:25:34.013692Z","signature_b64":"Ynz+x+YaGAcL5fUdY/xgxdZReCKHpa+EbpEBQKexbM3SkbCcNko7dPswDJSVluRa3q6UQ7dygil2wLoKeHQBAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8cf6bdb83e59a5f50f1023fa6ad5c7fc250794a103a210ca205d94b1dc53a9ab","last_reissued_at":"2026-07-05T07:25:34.013159Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:25:34.013159Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tuning LayerNorm in Attention: Towards Efficient Multi-Modal LLM Finetuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Bingchen Zhao, Chen Wei, Cihang Xie, Haoqin Tu, Jieru Mei","submitted_at":"2023-12-18T18:21:43Z","abstract_excerpt":"This paper introduces an efficient strategy to transform Large Language Models (LLMs) into Multi-Modal Large Language Models (MLLMs). By conceptualizing this transformation as a domain adaptation process, i.e., transitioning from text understanding to embracing multiple modalities, we intriguingly note that, within each attention block, tuning LayerNorm suffices to yield strong performance. Moreover, when benchmarked against other tuning approaches like full parameter finetuning or LoRA, its benefits on efficiency are substantial. For example, when compared to LoRA on a 13B model scale, perfor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.11420","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.11420/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.11420","created_at":"2026-07-05T07:25:34.013218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.11420v1","created_at":"2026-07-05T07:25:34.013218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.11420","created_at":"2026-07-05T07:25:34.013218+00:00"},{"alias_kind":"pith_short_12","alias_value":"RT3L3OB6LGS7","created_at":"2026-07-05T07:25:34.013218+00:00"},{"alias_kind":"pith_short_16","alias_value":"RT3L3OB6LGS7KDYQ","created_at":"2026-07-05T07:25:34.013218+00:00"},{"alias_kind":"pith_short_8","alias_value":"RT3L3OB6","created_at":"2026-07-05T07:25:34.013218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.11237","citing_title":"Concept Drift Guided LayerNorm Tuning for Efficient Multimodal Metaphor Identification","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14608","citing_title":"Parameter-Efficient Fine-Tuning for Large Models: A Comprehensive Survey","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12006","citing_title":"Robust Promptable Video Object Segmentation","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q","json":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q.json","graph_json":"https://pith.science/api/pith-number/RT3L3OB6LGS7KDYQEP5GVVOH7Q/graph.json","events_json":"https://pith.science/api/pith-number/RT3L3OB6LGS7KDYQEP5GVVOH7Q/events.json","paper":"https://pith.science/paper/RT3L3OB6"},"agent_actions":{"view_html":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q","download_json":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q.json","view_paper":"https://pith.science/paper/RT3L3OB6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.11420&json=true","fetch_graph":"https://pith.science/api/pith-number/RT3L3OB6LGS7KDYQEP5GVVOH7Q/graph.json","fetch_events":"https://pith.science/api/pith-number/RT3L3OB6LGS7KDYQEP5GVVOH7Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q/action/storage_attestation","attest_author":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q/action/author_attestation","sign_citation":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q/action/citation_signature","submit_replication":"https://pith.science/pith/RT3L3OB6LGS7KDYQEP5GVVOH7Q/action/replication_record"}},"created_at":"2026-07-05T07:25:34.013218+00:00","updated_at":"2026-07-05T07:25:34.013218+00:00"}