{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:JK22FA35JPYQMROVOB4QEK7VXW","short_pith_number":"pith:JK22FA35","schema_version":"1.0","canonical_sha256":"4ab5a2837d4bf10645d57079022bf5bd9b8a7cb6f4ac1708e2226fb34ab8ceab","source":{"kind":"arxiv","id":"1906.05890","version":4},"attestation_state":"computed","paper":{"title":"Gradient Descent Maximizes the Margin of Homogeneous Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jian Li, Kaifeng Lyu","submitted_at":"2019-06-13T18:52:00Z","abstract_excerpt":"In this paper, we study the implicit regularization of the gradient descent algorithm in homogeneous neural networks, including fully-connected and convolutional neural networks with ReLU or LeakyReLU activations. In particular, we study the gradient descent or gradient flow (i.e., gradient descent with infinitesimal step size) optimizing the logistic loss or cross-entropy loss of any homogeneous model (possibly non-smooth), and show that if the training loss decreases below a certain threshold, then we can define a smoothed version of the normalized margin which increases over time. We also f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1906.05890","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-06-13T18:52:00Z","cross_cats_sorted":["cs.NE","stat.ML"],"title_canon_sha256":"57ca1616bb2e236f8db27eaae033accb8851cc3c28db415e50ab6dcfc8d50fa7","abstract_canon_sha256":"ea6de9a6d0d3a8a93eab9fcb8eaf75144cc45f53c8c3553f8face3116abc2dcd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:02:50.426054Z","signature_b64":"oMOxMWQAqhuXSWL4NBSjbotZ6CCkLyW5plh9++RVK1SHEnubCMwoBoshGrtlv0x1SrM/Lwvgvy59x2qF9+WUCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ab5a2837d4bf10645d57079022bf5bd9b8a7cb6f4ac1708e2226fb34ab8ceab","last_reissued_at":"2026-07-05T02:02:50.425587Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:02:50.425587Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Gradient Descent Maximizes the Margin of Homogeneous Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jian Li, Kaifeng Lyu","submitted_at":"2019-06-13T18:52:00Z","abstract_excerpt":"In this paper, we study the implicit regularization of the gradient descent algorithm in homogeneous neural networks, including fully-connected and convolutional neural networks with ReLU or LeakyReLU activations. In particular, we study the gradient descent or gradient flow (i.e., gradient descent with infinitesimal step size) optimizing the logistic loss or cross-entropy loss of any homogeneous model (possibly non-smooth), and show that if the training loss decreases below a certain threshold, then we can define a smoothed version of the normalized margin which increases over time. We also f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1906.05890","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1906.05890/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1906.05890","created_at":"2026-07-05T02:02:50.425650+00:00"},{"alias_kind":"arxiv_version","alias_value":"1906.05890v4","created_at":"2026-07-05T02:02:50.425650+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1906.05890","created_at":"2026-07-05T02:02:50.425650+00:00"},{"alias_kind":"pith_short_12","alias_value":"JK22FA35JPYQ","created_at":"2026-07-05T02:02:50.425650+00:00"},{"alias_kind":"pith_short_16","alias_value":"JK22FA35JPYQMROV","created_at":"2026-07-05T02:02:50.425650+00:00"},{"alias_kind":"pith_short_8","alias_value":"JK22FA35","created_at":"2026-07-05T02:02:50.425650+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18465","citing_title":"What Does the Weight Norm Control in Grokking? Logit-Scale Mediation under Cross-Entropy","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10089","citing_title":"A Theory on Flow Matching with Neural Networks","ref_index":262,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17606","citing_title":"The Neural Tangent Kernel for Classification","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30559","citing_title":"Convergence of Continual Learning in Homogeneous Deep Networks","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23087","citing_title":"The Implicit Bias of Depth: From Neural Collapse to Softmax Codes","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17606","citing_title":"The Neural Tangent Kernel for Classification","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.09290","citing_title":"Prediction horizon shapes representations in predictive learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01642","citing_title":"The Effect of Mini-Batch Noise on the Implicit Bias of Adam","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02622","citing_title":"Implicit Bias in Deep Linear Discriminant Analysis","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2401.01335","citing_title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","ref_index":243,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06519","citing_title":"Efficient Techniques for Data Reconstruction, with Finite-Width Recovery Guarantees","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW","json":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW.json","graph_json":"https://pith.science/api/pith-number/JK22FA35JPYQMROVOB4QEK7VXW/graph.json","events_json":"https://pith.science/api/pith-number/JK22FA35JPYQMROVOB4QEK7VXW/events.json","paper":"https://pith.science/paper/JK22FA35"},"agent_actions":{"view_html":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW","download_json":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW.json","view_paper":"https://pith.science/paper/JK22FA35","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1906.05890&json=true","fetch_graph":"https://pith.science/api/pith-number/JK22FA35JPYQMROVOB4QEK7VXW/graph.json","fetch_events":"https://pith.science/api/pith-number/JK22FA35JPYQMROVOB4QEK7VXW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW/action/storage_attestation","attest_author":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW/action/author_attestation","sign_citation":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW/action/citation_signature","submit_replication":"https://pith.science/pith/JK22FA35JPYQMROVOB4QEK7VXW/action/replication_record"}},"created_at":"2026-07-05T02:02:50.425650+00:00","updated_at":"2026-07-05T02:02:50.425650+00:00"}