{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SWXJ6HLWLGWS7CZURX2BYPA4PQ","short_pith_number":"pith:SWXJ6HLW","schema_version":"1.0","canonical_sha256":"95ae9f1d7659ad2f8b348df41c3c1c7c122ce4f0bfaa2f9cb7f9d097a48979d3","source":{"kind":"arxiv","id":"2303.17612","version":3},"attestation_state":"computed","paper":{"title":"oBERTa: Improving Sparse Transfer Learning via improved initialization, distillation, and pruning regimes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandre Marques, ChengXiang Zhai, Daniel Campos, Mark Kurtz","submitted_at":"2023-03-30T01:37:19Z","abstract_excerpt":"In this paper, we introduce the range of oBERTa language models, an easy-to-use set of language models which allows Natural Language Processing (NLP) practitioners to obtain between 3.8 and 24.3 times faster models without expertise in model compression. Specifically, oBERTa extends existing work on pruning, knowledge distillation, and quantization and leverages frozen embeddings improves distillation and model initialization to deliver higher accuracy on a broad range of transfer tasks. In generating oBERTa, we explore how the highly optimized RoBERTa differs from the BERT for pruning during "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.17612","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-30T01:37:19Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3e2cb2cd2ac0b72c6da1d9513a21eacc778f85056d79cbec730092760ae753de","abstract_canon_sha256":"861422217606452741a75d7d0486a751b4111c560a2b22d523ffa97143b4aead"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:49.399809Z","signature_b64":"IN8CK/YAIhOhW4ON+rKgocDr33syOPtCkqRKdmVvX1EPMHkfaeEA/k9qwyWm3V9PQBEokaTsFc3F4auWKxWyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95ae9f1d7659ad2f8b348df41c3c1c7c122ce4f0bfaa2f9cb7f9d097a48979d3","last_reissued_at":"2026-07-05T06:17:49.399349Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:49.399349Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"oBERTa: Improving Sparse Transfer Learning via improved initialization, distillation, and pruning regimes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandre Marques, ChengXiang Zhai, Daniel Campos, Mark Kurtz","submitted_at":"2023-03-30T01:37:19Z","abstract_excerpt":"In this paper, we introduce the range of oBERTa language models, an easy-to-use set of language models which allows Natural Language Processing (NLP) practitioners to obtain between 3.8 and 24.3 times faster models without expertise in model compression. Specifically, oBERTa extends existing work on pruning, knowledge distillation, and quantization and leverages frozen embeddings improves distillation and model initialization to deliver higher accuracy on a broad range of transfer tasks. In generating oBERTa, we explore how the highly optimized RoBERTa differs from the BERT for pruning during "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.17612","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.17612/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.17612","created_at":"2026-07-05T06:17:49.399414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.17612v3","created_at":"2026-07-05T06:17:49.399414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.17612","created_at":"2026-07-05T06:17:49.399414+00:00"},{"alias_kind":"pith_short_12","alias_value":"SWXJ6HLWLGWS","created_at":"2026-07-05T06:17:49.399414+00:00"},{"alias_kind":"pith_short_16","alias_value":"SWXJ6HLWLGWS7CZU","created_at":"2026-07-05T06:17:49.399414+00:00"},{"alias_kind":"pith_short_8","alias_value":"SWXJ6HLW","created_at":"2026-07-05T06:17:49.399414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.01514","citing_title":"MeVe: A Modular System for Memory Verification and Effective Context Control in Language Models","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ","json":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ.json","graph_json":"https://pith.science/api/pith-number/SWXJ6HLWLGWS7CZURX2BYPA4PQ/graph.json","events_json":"https://pith.science/api/pith-number/SWXJ6HLWLGWS7CZURX2BYPA4PQ/events.json","paper":"https://pith.science/paper/SWXJ6HLW"},"agent_actions":{"view_html":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ","download_json":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ.json","view_paper":"https://pith.science/paper/SWXJ6HLW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.17612&json=true","fetch_graph":"https://pith.science/api/pith-number/SWXJ6HLWLGWS7CZURX2BYPA4PQ/graph.json","fetch_events":"https://pith.science/api/pith-number/SWXJ6HLWLGWS7CZURX2BYPA4PQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ/action/storage_attestation","attest_author":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ/action/author_attestation","sign_citation":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ/action/citation_signature","submit_replication":"https://pith.science/pith/SWXJ6HLWLGWS7CZURX2BYPA4PQ/action/replication_record"}},"created_at":"2026-07-05T06:17:49.399414+00:00","updated_at":"2026-07-05T06:17:49.399414+00:00"}