{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:M7HNUOVLLK4NNGAOTEHYZCNVYI","short_pith_number":"pith:M7HNUOVL","schema_version":"1.0","canonical_sha256":"67ceda3aab5ab8d6980e990f8c89b5c2221f87150abb894ba9fd29e9475c3754","source":{"kind":"arxiv","id":"2004.12406","version":2},"attestation_state":"computed","paper":{"title":"Masking as an Efficient Alternative to Finetuning for Pretrained Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fei Mi, Hinrich Sch\\\"utze, Martin Jaggi, Mengjie Zhao, Tao Lin","submitted_at":"2020-04-26T15:03:47Z","abstract_excerpt":"We present an efficient method of utilizing pretrained language models, where we learn selective binary masks for pretrained weights in lieu of modifying them through finetuning. Extensive evaluations of masking BERT and RoBERTa on a series of NLP tasks show that our masking scheme yields performance comparable to finetuning, yet has a much smaller memory footprint when several tasks need to be inferred simultaneously. Through intrinsic evaluations, we show that representations computed by masked language models encode information necessary for solving downstream tasks. Analyzing the loss land"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.12406","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2020-04-26T15:03:47Z","cross_cats_sorted":[],"title_canon_sha256":"fedeb3ea9966e45cf751587f97615a23b38085630ad983c1e08cce3624edda5b","abstract_canon_sha256":"1dc53766d9a99fa6b254c1ff76cda85188ff95f4d7d033e8584bae2087603ea1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:41:53.270519Z","signature_b64":"LtNRx2YiiJinfBeT2h6rlWI24vKqmWZn1tSnIgWS/ovri7a4ASPfQ4XwjghEpnWKsTIuNViHiNTLnKAMyXjdDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"67ceda3aab5ab8d6980e990f8c89b5c2221f87150abb894ba9fd29e9475c3754","last_reissued_at":"2026-07-05T01:41:53.270117Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:41:53.270117Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Masking as an Efficient Alternative to Finetuning for Pretrained Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fei Mi, Hinrich Sch\\\"utze, Martin Jaggi, Mengjie Zhao, Tao Lin","submitted_at":"2020-04-26T15:03:47Z","abstract_excerpt":"We present an efficient method of utilizing pretrained language models, where we learn selective binary masks for pretrained weights in lieu of modifying them through finetuning. Extensive evaluations of masking BERT and RoBERTa on a series of NLP tasks show that our masking scheme yields performance comparable to finetuning, yet has a much smaller memory footprint when several tasks need to be inferred simultaneously. Through intrinsic evaluations, we show that representations computed by masked language models encode information necessary for solving downstream tasks. Analyzing the loss land"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.12406","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.12406/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.12406","created_at":"2026-07-05T01:41:53.270181+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.12406v2","created_at":"2026-07-05T01:41:53.270181+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.12406","created_at":"2026-07-05T01:41:53.270181+00:00"},{"alias_kind":"pith_short_12","alias_value":"M7HNUOVLLK4N","created_at":"2026-07-05T01:41:53.270181+00:00"},{"alias_kind":"pith_short_16","alias_value":"M7HNUOVLLK4NNGAO","created_at":"2026-07-05T01:41:53.270181+00:00"},{"alias_kind":"pith_short_8","alias_value":"M7HNUOVL","created_at":"2026-07-05T01:41:53.270181+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.19219","citing_title":"Selective LoRA for Visual Tokens and Attention Heads","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2101.00190","citing_title":"Prefix-Tuning: Optimizing Continuous Prompts for Generation","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI","json":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI.json","graph_json":"https://pith.science/api/pith-number/M7HNUOVLLK4NNGAOTEHYZCNVYI/graph.json","events_json":"https://pith.science/api/pith-number/M7HNUOVLLK4NNGAOTEHYZCNVYI/events.json","paper":"https://pith.science/paper/M7HNUOVL"},"agent_actions":{"view_html":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI","download_json":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI.json","view_paper":"https://pith.science/paper/M7HNUOVL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.12406&json=true","fetch_graph":"https://pith.science/api/pith-number/M7HNUOVLLK4NNGAOTEHYZCNVYI/graph.json","fetch_events":"https://pith.science/api/pith-number/M7HNUOVLLK4NNGAOTEHYZCNVYI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI/action/storage_attestation","attest_author":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI/action/author_attestation","sign_citation":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI/action/citation_signature","submit_replication":"https://pith.science/pith/M7HNUOVLLK4NNGAOTEHYZCNVYI/action/replication_record"}},"created_at":"2026-07-05T01:41:53.270181+00:00","updated_at":"2026-07-05T01:41:53.270181+00:00"}