{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NPZCHRX3V67INE4BMOP2SI3JEP","short_pith_number":"pith:NPZCHRX3","schema_version":"1.0","canonical_sha256":"6bf223c6fbafbe869381639fa9236923f9ac79cbb98f70d2cb8cb541ed6a630a","source":{"kind":"arxiv","id":"2402.17193","version":1},"attestation_state":"computed","paper":{"title":"When Scaling Meets LLM Finetuning: The Effect of Data, Model and Finetuning Method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Biao Zhang, Colin Cherry, Orhan Firat, Zhongtao Liu","submitted_at":"2024-02-27T04:18:49Z","abstract_excerpt":"While large language models (LLMs) often adopt finetuning to unlock their capabilities for downstream applications, our understanding on the inductive biases (especially the scaling properties) of different finetuning methods is still limited. To fill this gap, we conduct systematic experiments studying whether and how different scaling factors, including LLM model size, pretraining data size, new finetuning parameter size and finetuning data size, affect the finetuning performance. We consider two types of finetuning -- full-model tuning (FMT) and parameter efficient tuning (PET, including pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17193","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-27T04:18:49Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"96d1101b7e4c77d4c4c9e724e67ff071deb7dfd721bff1db6c07cc87afb9d447","abstract_canon_sha256":"16d9cea7432e398b5c98272834dcdfd27bd1b0728f3f26a2a1e8d617a459d6a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:43.864348Z","signature_b64":"fhzLFlpB+SpeQOZIo6q9En8Im9oKTeeNdTOmXu730KtJmdEYX9oRZBSKDi6JRaVMWFB0opx2clnTj6JPS9QwDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6bf223c6fbafbe869381639fa9236923f9ac79cbb98f70d2cb8cb541ed6a630a","last_reissued_at":"2026-07-05T07:49:43.863867Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:43.863867Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Scaling Meets LLM Finetuning: The Effect of Data, Model and Finetuning Method","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Biao Zhang, Colin Cherry, Orhan Firat, Zhongtao Liu","submitted_at":"2024-02-27T04:18:49Z","abstract_excerpt":"While large language models (LLMs) often adopt finetuning to unlock their capabilities for downstream applications, our understanding on the inductive biases (especially the scaling properties) of different finetuning methods is still limited. To fill this gap, we conduct systematic experiments studying whether and how different scaling factors, including LLM model size, pretraining data size, new finetuning parameter size and finetuning data size, affect the finetuning performance. We consider two types of finetuning -- full-model tuning (FMT) and parameter efficient tuning (PET, including pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17193","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17193/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17193","created_at":"2026-07-05T07:49:43.863921+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17193v1","created_at":"2026-07-05T07:49:43.863921+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17193","created_at":"2026-07-05T07:49:43.863921+00:00"},{"alias_kind":"pith_short_12","alias_value":"NPZCHRX3V67I","created_at":"2026-07-05T07:49:43.863921+00:00"},{"alias_kind":"pith_short_16","alias_value":"NPZCHRX3V67INE4B","created_at":"2026-07-05T07:49:43.863921+00:00"},{"alias_kind":"pith_short_8","alias_value":"NPZCHRX3","created_at":"2026-07-05T07:49:43.863921+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24119","citing_title":"When Top-1 Fails: Calibrating LoRA Monitors for Masked Diffusion LMs","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30360","citing_title":"On the Vulnerability of Parameter-Level Defenses to Model Merging","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2505.01307","citing_title":"Document Retrieval Augmented Fine-Tuning (DRAFT) for safety-critical software assessments","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21127","citing_title":"Reasoning-Trace Collapse: Evaluating the Loss of Explicit Reasoning During Fine-Tuning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2506.11563","citing_title":"A Survey of Personalized Federated Foundation Models for Privacy-Preserving Recommendation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05929","citing_title":"LLM Harms: A Taxonomy and Discussion","ref_index":212,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00989","citing_title":"Sustainable Code Generation Using Large Language Models: A Systematic Literature Review","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06202","citing_title":"Cross-Lingual Transfer and Parameter-Efficient Adaptation in the Turkic Language Family: A Theoretical Framework for Low-Resource Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18381","citing_title":"Learning from Less: Measuring the Effectiveness of RLVR in Low Data and Compute Regimes","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15270","citing_title":"Enhancing Large Language Models with Retrieval Augmented Generation for Software Testing and Inspection Automation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17267","citing_title":"Rectification Difficulty and Optimal Sample Allocation in LLM-Augmented Surveys","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17778","citing_title":"TeleEmbedBench: A Multi-Corpus Embedding Benchmark for RAG in Telecommunications","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20924","citing_title":"Clinically Interpretable Sepsis Early Warning via LLM-Guided Simulation of Temporal Physiological Dynamics","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP","json":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP.json","graph_json":"https://pith.science/api/pith-number/NPZCHRX3V67INE4BMOP2SI3JEP/graph.json","events_json":"https://pith.science/api/pith-number/NPZCHRX3V67INE4BMOP2SI3JEP/events.json","paper":"https://pith.science/paper/NPZCHRX3"},"agent_actions":{"view_html":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP","download_json":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP.json","view_paper":"https://pith.science/paper/NPZCHRX3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17193&json=true","fetch_graph":"https://pith.science/api/pith-number/NPZCHRX3V67INE4BMOP2SI3JEP/graph.json","fetch_events":"https://pith.science/api/pith-number/NPZCHRX3V67INE4BMOP2SI3JEP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP/action/storage_attestation","attest_author":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP/action/author_attestation","sign_citation":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP/action/citation_signature","submit_replication":"https://pith.science/pith/NPZCHRX3V67INE4BMOP2SI3JEP/action/replication_record"}},"created_at":"2026-07-05T07:49:43.863921+00:00","updated_at":"2026-07-05T07:49:43.863921+00:00"}