{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DOB7M6U6I7LNX6BDS4ZM74GNL2","short_pith_number":"pith:DOB7M6U6","schema_version":"1.0","canonical_sha256":"1b83f67a9e47d6dbf8239732cff0cd5eb5ea64b1cd81a907f647964c71e2b166","source":{"kind":"arxiv","id":"2502.13811","version":2},"attestation_state":"computed","paper":{"title":"On the Duality between Gradient Transformations and Adapters","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Han Guo, Hunter Lang, Lucas Torroba-Hennigen, Yoon Kim","submitted_at":"2025-02-19T15:26:18Z","abstract_excerpt":"We study memory-efficient optimization of neural networks (in particular language models) with linear gradient transformations, where the gradients are linearly mapped to a lower dimensional space than the full parameter space, thus saving memory required for gradient accumulation and optimizer state persistence. The model parameters are updated by first performing an optimization step in the lower dimensional space and then going back into the original parameter space via the linear map's transpose. We show that optimizing the model in this transformed space is equivalent to reparameterizing "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.13811","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-19T15:26:18Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ab9a7c0d4916cdabdb28eff014904c843c6685b1233bbfb7a905f905f62f477d","abstract_canon_sha256":"35384352e3ec6cacc35f8ac7ee45b4d3a035b603e7bdeaf6ea190a418190bdd5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:51:22.194272Z","signature_b64":"dHavncHMt3FmeEpBSl8Zn6msxR/zSNS2JJccorH000QxY/irkIWElDsPMju/NeoiXyYmrjhEJzd3ot2D/xf7CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b83f67a9e47d6dbf8239732cff0cd5eb5ea64b1cd81a907f647964c71e2b166","last_reissued_at":"2026-07-05T11:51:22.193752Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:51:22.193752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Duality between Gradient Transformations and Adapters","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Han Guo, Hunter Lang, Lucas Torroba-Hennigen, Yoon Kim","submitted_at":"2025-02-19T15:26:18Z","abstract_excerpt":"We study memory-efficient optimization of neural networks (in particular language models) with linear gradient transformations, where the gradients are linearly mapped to a lower dimensional space than the full parameter space, thus saving memory required for gradient accumulation and optimizer state persistence. The model parameters are updated by first performing an optimization step in the lower dimensional space and then going back into the original parameter space via the linear map's transpose. We show that optimizing the model in this transformed space is equivalent to reparameterizing "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13811","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.13811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.13811","created_at":"2026-07-05T11:51:22.193823+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.13811v2","created_at":"2026-07-05T11:51:22.193823+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13811","created_at":"2026-07-05T11:51:22.193823+00:00"},{"alias_kind":"pith_short_12","alias_value":"DOB7M6U6I7LN","created_at":"2026-07-05T11:51:22.193823+00:00"},{"alias_kind":"pith_short_16","alias_value":"DOB7M6U6I7LNX6BD","created_at":"2026-07-05T11:51:22.193823+00:00"},{"alias_kind":"pith_short_8","alias_value":"DOB7M6U6","created_at":"2026-07-05T11:51:22.193823+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06806","citing_title":"Logits are All We Need to Adapt Closed Models","ref_index":56,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2","json":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2.json","graph_json":"https://pith.science/api/pith-number/DOB7M6U6I7LNX6BDS4ZM74GNL2/graph.json","events_json":"https://pith.science/api/pith-number/DOB7M6U6I7LNX6BDS4ZM74GNL2/events.json","paper":"https://pith.science/paper/DOB7M6U6"},"agent_actions":{"view_html":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2","download_json":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2.json","view_paper":"https://pith.science/paper/DOB7M6U6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.13811&json=true","fetch_graph":"https://pith.science/api/pith-number/DOB7M6U6I7LNX6BDS4ZM74GNL2/graph.json","fetch_events":"https://pith.science/api/pith-number/DOB7M6U6I7LNX6BDS4ZM74GNL2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2/action/storage_attestation","attest_author":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2/action/author_attestation","sign_citation":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2/action/citation_signature","submit_replication":"https://pith.science/pith/DOB7M6U6I7LNX6BDS4ZM74GNL2/action/replication_record"}},"created_at":"2026-07-05T11:51:22.193823+00:00","updated_at":"2026-07-05T11:51:22.193823+00:00"}