{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WZEOCBI7CC7BGPSR7VC3H7FBCQ","short_pith_number":"pith:WZEOCBI7","schema_version":"1.0","canonical_sha256":"b648e1051f10be133e51fd45b3fca11436ad71df167117d994c7f8d2da3fa331","source":{"kind":"arxiv","id":"2307.05695","version":4},"attestation_state":"computed","paper":{"title":"ReLoRA: High-Rank Training Through Low-Rank Updates","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anna Rumshisky, Namrata Shivagunde, Sherin Muckatira, Vladislav Lialin","submitted_at":"2023-07-11T18:02:09Z","abstract_excerpt":"Despite the dominance and effectiveness of scaling, resulting in large networks with hundreds of billions of parameters, the necessity to train overparameterized models remains poorly understood, while training costs grow exponentially. In this paper, we explore parameter-efficient training techniques as an approach to training large neural networks. We introduce a novel method called ReLoRA, which utilizes low-rank updates to train high-rank networks. We apply ReLoRA to training transformer language models with up to 1.3B parameters and demonstrate comparable performance to regular neural net"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.05695","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-07-11T18:02:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"32deef60090e94a8368c9bff0e073062433882850df9c38e2816f9f89a8b8035","abstract_canon_sha256":"762ce6d0602ed59484bc4ef901fdb858c6355f9b971982f41622aabc16721317"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:22:20.758639Z","signature_b64":"cQ1Xs3RHxzYuGv2QiAPfnK1q4e6jgCVGlfwqQ8uXo+z0np0zXQRuNRj/BX2tAxEk3vtIHx0T/5qUKkY9tym+CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b648e1051f10be133e51fd45b3fca11436ad71df167117d994c7f8d2da3fa331","last_reissued_at":"2026-07-05T07:22:20.758169Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:22:20.758169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReLoRA: High-Rank Training Through Low-Rank Updates","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anna Rumshisky, Namrata Shivagunde, Sherin Muckatira, Vladislav Lialin","submitted_at":"2023-07-11T18:02:09Z","abstract_excerpt":"Despite the dominance and effectiveness of scaling, resulting in large networks with hundreds of billions of parameters, the necessity to train overparameterized models remains poorly understood, while training costs grow exponentially. In this paper, we explore parameter-efficient training techniques as an approach to training large neural networks. We introduce a novel method called ReLoRA, which utilizes low-rank updates to train high-rank networks. We apply ReLoRA to training transformer language models with up to 1.3B parameters and demonstrate comparable performance to regular neural net"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.05695","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.05695/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.05695","created_at":"2026-07-05T07:22:20.758224+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.05695v4","created_at":"2026-07-05T07:22:20.758224+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.05695","created_at":"2026-07-05T07:22:20.758224+00:00"},{"alias_kind":"pith_short_12","alias_value":"WZEOCBI7CC7B","created_at":"2026-07-05T07:22:20.758224+00:00"},{"alias_kind":"pith_short_16","alias_value":"WZEOCBI7CC7BGPSR","created_at":"2026-07-05T07:22:20.758224+00:00"},{"alias_kind":"pith_short_8","alias_value":"WZEOCBI7","created_at":"2026-07-05T07:22:20.758224+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08754","citing_title":"SLORR: Simple and Efficient In-Training Low-Rank Regularization","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28932","citing_title":"DLR: Zero-Inference-Cost Latent Residuals for Low-Rank Pre-Training","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29184","citing_title":"BaRA: Bayesian Adaptive Rank Allocation for Parameter-Efficient Fine-Tuning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21147","citing_title":"SMoA: Spectrum Modulation Adapter for Parameter-Efficient Fine-Tuning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18541","citing_title":"LESSViT: Robust Hyperspectral Representation Learning under Spectral Configuration Shift","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18993","citing_title":"CR-Net: Scaling Parameter-Efficient Training with Cross-Layer Low-Rank Structure","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12131","citing_title":"BOOST: BOttleneck-Optimized Scalable Training Framework for Low-Rank Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27308","citing_title":"BoostLoRA: Growing Effective Rank by Boosting Adapters","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04079","citing_title":"Efficient Handwriting-Based Alzheimer,s Disease Diagnosis Using a Low-Rank Mixture of Experts Deep Learning Framework","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ","json":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ.json","graph_json":"https://pith.science/api/pith-number/WZEOCBI7CC7BGPSR7VC3H7FBCQ/graph.json","events_json":"https://pith.science/api/pith-number/WZEOCBI7CC7BGPSR7VC3H7FBCQ/events.json","paper":"https://pith.science/paper/WZEOCBI7"},"agent_actions":{"view_html":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ","download_json":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ.json","view_paper":"https://pith.science/paper/WZEOCBI7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.05695&json=true","fetch_graph":"https://pith.science/api/pith-number/WZEOCBI7CC7BGPSR7VC3H7FBCQ/graph.json","fetch_events":"https://pith.science/api/pith-number/WZEOCBI7CC7BGPSR7VC3H7FBCQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ/action/storage_attestation","attest_author":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ/action/author_attestation","sign_citation":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ/action/citation_signature","submit_replication":"https://pith.science/pith/WZEOCBI7CC7BGPSR7VC3H7FBCQ/action/replication_record"}},"created_at":"2026-07-05T07:22:20.758224+00:00","updated_at":"2026-07-05T07:22:20.758224+00:00"}