{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:PDT6RFZRA7J3VJKJP4FUZ3WENZ","short_pith_number":"pith:PDT6RFZR","schema_version":"1.0","canonical_sha256":"78e7e8973107d3baa5497f0b4ceec46e75697f798951c82e9dca44f8bfebe8de","source":{"kind":"arxiv","id":"2002.06305","version":1},"attestation_state":"computed","paper":{"title":"Fine-Tuning Pretrained Language Models: Weight Initializations, Data Orders, and Early Stopping","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ali Farhadi, Gabriel Ilharco, Hannaneh Hajishirzi, Jesse Dodge, Noah Smith, Roy Schwartz","submitted_at":"2020-02-15T02:40:10Z","abstract_excerpt":"Fine-tuning pretrained contextual word embedding models to supervised downstream tasks has become commonplace in natural language processing. This process, however, is often brittle: even with the same hyperparameter values, distinct random seeds can lead to substantially different results. To better understand this phenomenon, we experiment with four datasets from the GLUE benchmark, fine-tuning BERT hundreds of times on each while varying only the random seeds. We find substantial performance increases compared to previously reported results, and we quantify how the performance of the best-f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.06305","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-02-15T02:40:10Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"43b2fa36f1978736731ce83c0e7d9a2de74dd2671d569b304c0456d89a4888fc","abstract_canon_sha256":"689a3f95c0324212d060efe1020f1afa67e20276232584334357d6d2204c9f11"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:42:00.524661Z","signature_b64":"zZYROHzU7R+wC+gjENblSrWr1nPxdi/QpcovhZgGdnaNt87clvQRwh9TAjfoRTmM2HI1zGb21q4d5q1YsUf6DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78e7e8973107d3baa5497f0b4ceec46e75697f798951c82e9dca44f8bfebe8de","last_reissued_at":"2026-07-05T00:42:00.524184Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:42:00.524184Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-Tuning Pretrained Language Models: Weight Initializations, Data Orders, and Early Stopping","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ali Farhadi, Gabriel Ilharco, Hannaneh Hajishirzi, Jesse Dodge, Noah Smith, Roy Schwartz","submitted_at":"2020-02-15T02:40:10Z","abstract_excerpt":"Fine-tuning pretrained contextual word embedding models to supervised downstream tasks has become commonplace in natural language processing. This process, however, is often brittle: even with the same hyperparameter values, distinct random seeds can lead to substantially different results. To better understand this phenomenon, we experiment with four datasets from the GLUE benchmark, fine-tuning BERT hundreds of times on each while varying only the random seeds. We find substantial performance increases compared to previously reported results, and we quantify how the performance of the best-f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.06305","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.06305/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.06305","created_at":"2026-07-05T00:42:00.524238+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.06305v1","created_at":"2026-07-05T00:42:00.524238+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.06305","created_at":"2026-07-05T00:42:00.524238+00:00"},{"alias_kind":"pith_short_12","alias_value":"PDT6RFZRA7J3","created_at":"2026-07-05T00:42:00.524238+00:00"},{"alias_kind":"pith_short_16","alias_value":"PDT6RFZRA7J3VJKJ","created_at":"2026-07-05T00:42:00.524238+00:00"},{"alias_kind":"pith_short_8","alias_value":"PDT6RFZR","created_at":"2026-07-05T00:42:00.524238+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":30,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22917","citing_title":"GRAIN: Group Aggregation via Min-Norm Objective","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22606","citing_title":"Sub-Billion, Super-Frontier: Small Language Models Rival Zero-Shot Frontier LLMs on General and Literary Relation Extraction","ref_index":249,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19988","citing_title":"Repository-Level Solidity Code Generation with Large Language Models: From Prompting to Fine-Tuning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20536","citing_title":"The FID Lottery: Quantifying Hidden Randomness in Generative-Model Evaluation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19638","citing_title":"MiqraBERT: Regression-Based Sentence-BERT Finetuning for Biblical Hebrew Parallel Detection","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18521","citing_title":"Sparsity Curse: Understanding RLVR Model Parameter Space from Model Merging","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07082","citing_title":"On the Geometry of On-Policy Distillation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02100","citing_title":"PortBERT: Navigating the Depths of Portuguese Language Models","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31246","citing_title":"BadBone: Backdoor Attacks Against Backbone Models in Visual Prompt Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":272,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29554","citing_title":"Optimizer Memory Makes Shuffle Order a First-Order Source of Fine-Tuning Noise","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07559","citing_title":"Phantom Transitions in Language Model Fine-Tuning: A Density-Matrix Analysis","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29152","citing_title":"Do Deep Networks Forget Initialization? A Forgetting-Time View of Practical Inductive Bias","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19638","citing_title":"MiqraBERT: Regression-Based Sentence-BERT Finetuning for Biblical Hebrew Parallel Detection","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20052","citing_title":"PromptRad: Knowledge-Enhanced Multi-Label Prompt-Tuning for Low-Resource Radiology Report Labeling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19018","citing_title":"LoRA vs. Full Fine-Tuning: A Theoretical Perspective","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20052","citing_title":"PromptRad: Knowledge-Enhanced Multi-Label Prompt-Tuning for Low-Resource Radiology Report Labeling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00994","citing_title":"Should We Still Pretrain Encoders with Masked Language Modeling?","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2508.11196","citing_title":"UAV-VL-R1: Generalizing Vision-Language Models via Supervised Fine-Tuning and Multi-Stage GRPO for UAV Visual Reasoning","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2009.01325","citing_title":"Learning to summarize from human feedback","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2502.03387","citing_title":"LIMO: Less is More for Reasoning","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11405","citing_title":"20/20 Vision Language Models: A Prescription for Better VLMs through Data Curation Alone","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11206","citing_title":"Instructions Shape Production of Language, not Processing","ref_index":196,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02937","citing_title":"If It's Good Enough for You, It's Good Enough for Me: Transferability of Audio Sufficiencies across Models","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ","json":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ.json","graph_json":"https://pith.science/api/pith-number/PDT6RFZRA7J3VJKJP4FUZ3WENZ/graph.json","events_json":"https://pith.science/api/pith-number/PDT6RFZRA7J3VJKJP4FUZ3WENZ/events.json","paper":"https://pith.science/paper/PDT6RFZR"},"agent_actions":{"view_html":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ","download_json":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ.json","view_paper":"https://pith.science/paper/PDT6RFZR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.06305&json=true","fetch_graph":"https://pith.science/api/pith-number/PDT6RFZRA7J3VJKJP4FUZ3WENZ/graph.json","fetch_events":"https://pith.science/api/pith-number/PDT6RFZRA7J3VJKJP4FUZ3WENZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ/action/storage_attestation","attest_author":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ/action/author_attestation","sign_citation":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ/action/citation_signature","submit_replication":"https://pith.science/pith/PDT6RFZRA7J3VJKJP4FUZ3WENZ/action/replication_record"}},"created_at":"2026-07-05T00:42:00.524238+00:00","updated_at":"2026-07-05T00:42:00.524238+00:00"}