{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RWPO7EMKDPLOMNUZZXEKXUO5TI","short_pith_number":"pith:RWPO7EMK","schema_version":"1.0","canonical_sha256":"8d9eef918a1bd6e63699cdc8abd1dd9a2d6f376e6ddafc6e9c6f14c857dc8e45","source":{"kind":"arxiv","id":"2411.10696","version":1},"attestation_state":"computed","paper":{"title":"HELENE: Hessian Layer-wise Clipping and Gradient Annealing for Accelerating Fine-tuning LLM with Zeroth-order Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Fei Dou, Huaqin Zhao, Jiaxi Li, Jin Lu, Shizhe Liang, Tianming Liu, Wei Liu, Xiang Li, Xiaofeng Yang, Yi Pan","submitted_at":"2024-11-16T04:27:22Z","abstract_excerpt":"Fine-tuning large language models (LLMs) poses significant memory challenges, as the back-propagation process demands extensive resources, especially with growing model sizes. Recent work, MeZO, addresses this issue using a zeroth-order (ZO) optimization method, which reduces memory consumption by matching the usage to the inference phase. However, MeZO experiences slow convergence due to varying curvatures across model parameters. To overcome this limitation, we introduce HELENE, a novel scalable and memory-efficient optimizer that integrates annealed A-GNB gradients with a diagonal Hessian e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.10696","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-11-16T04:27:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bc9b27ab6561f82bf4c918dc15809114aba22710ef93f4d597a5fd1834063b5b","abstract_canon_sha256":"e2b2af21e56143b78a823712ecc4d777233ec28a5670f67a9d2a17a473f7e42c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:36:20.706324Z","signature_b64":"mafAP3/NnBkCnTBCNCg3x87XRsnbTvIHfNcS6CQfJT0tojUN+QN0KAyUJaaRQpKvSwezGJzfmx7HVlcyPV5FBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d9eef918a1bd6e63699cdc8abd1dd9a2d6f376e6ddafc6e9c6f14c857dc8e45","last_reissued_at":"2026-07-05T09:36:20.705815Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:36:20.705815Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HELENE: Hessian Layer-wise Clipping and Gradient Annealing for Accelerating Fine-tuning LLM with Zeroth-order Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Fei Dou, Huaqin Zhao, Jiaxi Li, Jin Lu, Shizhe Liang, Tianming Liu, Wei Liu, Xiang Li, Xiaofeng Yang, Yi Pan","submitted_at":"2024-11-16T04:27:22Z","abstract_excerpt":"Fine-tuning large language models (LLMs) poses significant memory challenges, as the back-propagation process demands extensive resources, especially with growing model sizes. Recent work, MeZO, addresses this issue using a zeroth-order (ZO) optimization method, which reduces memory consumption by matching the usage to the inference phase. However, MeZO experiences slow convergence due to varying curvatures across model parameters. To overcome this limitation, we introduce HELENE, a novel scalable and memory-efficient optimizer that integrates annealed A-GNB gradients with a diagonal Hessian e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.10696","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.10696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.10696","created_at":"2026-07-05T09:36:20.705873+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.10696v1","created_at":"2026-07-05T09:36:20.705873+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.10696","created_at":"2026-07-05T09:36:20.705873+00:00"},{"alias_kind":"pith_short_12","alias_value":"RWPO7EMKDPLO","created_at":"2026-07-05T09:36:20.705873+00:00"},{"alias_kind":"pith_short_16","alias_value":"RWPO7EMKDPLOMNUZ","created_at":"2026-07-05T09:36:20.705873+00:00"},{"alias_kind":"pith_short_8","alias_value":"RWPO7EMK","created_at":"2026-07-05T09:36:20.705873+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00650","citing_title":"AdaMeZO: Adam-style Zeroth-Order Optimizer for LLM Fine-tuning Without Maintaining the Moments","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI","json":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI.json","graph_json":"https://pith.science/api/pith-number/RWPO7EMKDPLOMNUZZXEKXUO5TI/graph.json","events_json":"https://pith.science/api/pith-number/RWPO7EMKDPLOMNUZZXEKXUO5TI/events.json","paper":"https://pith.science/paper/RWPO7EMK"},"agent_actions":{"view_html":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI","download_json":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI.json","view_paper":"https://pith.science/paper/RWPO7EMK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.10696&json=true","fetch_graph":"https://pith.science/api/pith-number/RWPO7EMKDPLOMNUZZXEKXUO5TI/graph.json","fetch_events":"https://pith.science/api/pith-number/RWPO7EMKDPLOMNUZZXEKXUO5TI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI/action/storage_attestation","attest_author":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI/action/author_attestation","sign_citation":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI/action/citation_signature","submit_replication":"https://pith.science/pith/RWPO7EMKDPLOMNUZZXEKXUO5TI/action/replication_record"}},"created_at":"2026-07-05T09:36:20.705873+00:00","updated_at":"2026-07-05T09:36:20.705873+00:00"}