{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:XBLQWJERKJGNAHUTXHYAW4O4FF","short_pith_number":"pith:XBLQWJER","schema_version":"1.0","canonical_sha256":"b8570b2491524cd01e93b9f00b71dc2948884855905bcd1a37fdf24ac08d9b40","source":{"kind":"arxiv","id":"1708.00523","version":8},"attestation_state":"computed","paper":{"title":"The duality structure gradient descent algorithm: analysis and applications to neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Thomas Flynn","submitted_at":"2017-08-01T21:24:38Z","abstract_excerpt":"The training of machine learning models is typically carried out using some form of gradient descent, often with great success. However, non-asymptotic analyses of first-order optimization algorithms typically employ a gradient smoothness assumption (formally, Lipschitz continuity of the gradient) that is too strong to be applicable in the case of deep neural networks. To address this, we propose an algorithm named duality structure gradient descent (DSGD) that is amenable to non-asymptotic performance analysis, under mild assumptions on the training set and network architecture. The algorithm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1708.00523","kind":"arxiv","version":8},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-08-01T21:24:38Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"7a18f790688b373d50140d38efa76afa4ef43405fda55ca963cffbcd27ee5225","abstract_canon_sha256":"d394bbd16023319624b196d43fa043bd525a3deb4fd01576c11e902635f9ac67"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:07.164552Z","signature_b64":"YuKRlUk9UyCc70jixpP/xG06ZeDX1xy0CiIkU1u9qWmJHVU8oQnI8VCUCg01nRAMkUK8EGDVN0gLZ7+cZFf/CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8570b2491524cd01e93b9f00b71dc2948884855905bcd1a37fdf24ac08d9b40","last_reissued_at":"2026-07-05T08:32:07.164047Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:07.164047Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The duality structure gradient descent algorithm: analysis and applications to neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Thomas Flynn","submitted_at":"2017-08-01T21:24:38Z","abstract_excerpt":"The training of machine learning models is typically carried out using some form of gradient descent, often with great success. However, non-asymptotic analyses of first-order optimization algorithms typically employ a gradient smoothness assumption (formally, Lipschitz continuity of the gradient) that is too strong to be applicable in the case of deep neural networks. To address this, we propose an algorithm named duality structure gradient descent (DSGD) that is amenable to non-asymptotic performance analysis, under mild assumptions on the training set and network architecture. The algorithm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1708.00523","kind":"arxiv","version":8},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1708.00523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1708.00523","created_at":"2026-07-05T08:32:07.164105+00:00"},{"alias_kind":"arxiv_version","alias_value":"1708.00523v8","created_at":"2026-07-05T08:32:07.164105+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1708.00523","created_at":"2026-07-05T08:32:07.164105+00:00"},{"alias_kind":"pith_short_12","alias_value":"XBLQWJERKJGN","created_at":"2026-07-05T08:32:07.164105+00:00"},{"alias_kind":"pith_short_16","alias_value":"XBLQWJERKJGNAHUT","created_at":"2026-07-05T08:32:07.164105+00:00"},{"alias_kind":"pith_short_8","alias_value":"XBLQWJER","created_at":"2026-07-05T08:32:07.164105+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.07529","citing_title":"Training Deep Learning Models with Norm-Constrained LMOs","ref_index":176,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17423","citing_title":"A unified convergence theory for adaptive first-order methods in the nonconvex case, including AdaNorm, full and diagonal AdaGrad, Shampoo and Muo","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF","json":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF.json","graph_json":"https://pith.science/api/pith-number/XBLQWJERKJGNAHUTXHYAW4O4FF/graph.json","events_json":"https://pith.science/api/pith-number/XBLQWJERKJGNAHUTXHYAW4O4FF/events.json","paper":"https://pith.science/paper/XBLQWJER"},"agent_actions":{"view_html":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF","download_json":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF.json","view_paper":"https://pith.science/paper/XBLQWJER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1708.00523&json=true","fetch_graph":"https://pith.science/api/pith-number/XBLQWJERKJGNAHUTXHYAW4O4FF/graph.json","fetch_events":"https://pith.science/api/pith-number/XBLQWJERKJGNAHUTXHYAW4O4FF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF/action/storage_attestation","attest_author":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF/action/author_attestation","sign_citation":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF/action/citation_signature","submit_replication":"https://pith.science/pith/XBLQWJERKJGNAHUTXHYAW4O4FF/action/replication_record"}},"created_at":"2026-07-05T08:32:07.164105+00:00","updated_at":"2026-07-05T08:32:07.164105+00:00"}