{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H3QG4QEGD6HVSURFE6YAEB2RDI","short_pith_number":"pith:H3QG4QEG","schema_version":"1.0","canonical_sha256":"3ee06e40861f8f59522527b00207511a07f0d6bb62b99052713140b02b226920","source":{"kind":"arxiv","id":"2407.10780","version":4},"attestation_state":"computed","paper":{"title":"Correlations Are Ruining Your Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Nasir Ahmad","submitted_at":"2024-07-15T14:59:43Z","abstract_excerpt":"Herein the topics of (natural) gradient descent, data decorrelation, and approximate methods for backpropagation are brought into a common discussion. Natural gradient descent illuminates how gradient vectors, pointing at directions of steepest descent, can be improved by considering the local curvature of loss landscapes. We extend this perspective and show that to fully solve the problem illuminated by natural gradients in neural networks, one must recognise that correlations in the data at any linear transformation, including node responses at every layer of a neural network, cause a non-or"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.10780","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-15T14:59:43Z","cross_cats_sorted":["cs.NE"],"title_canon_sha256":"0c8d54d77a6e0bb87eea468bcddb81a5ebfc4d567b0465a3674ef563b28bedd5","abstract_canon_sha256":"ff9d61d9d7ea8930bd371b50676b0dd2fce9b9583d95aa93ed1ee00fd1367fd1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:19.667393Z","signature_b64":"vc8GdTN+iK1zXOGnTsU0MxIb4pY6xEMkppyI+rdo6tJxe3hnoiqgKwZqBft8yDYQhjm3egn7IQ/H1DHVN8OnCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ee06e40861f8f59522527b00207511a07f0d6bb62b99052713140b02b226920","last_reissued_at":"2026-07-05T11:58:19.666941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:19.666941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Correlations Are Ruining Your Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Nasir Ahmad","submitted_at":"2024-07-15T14:59:43Z","abstract_excerpt":"Herein the topics of (natural) gradient descent, data decorrelation, and approximate methods for backpropagation are brought into a common discussion. Natural gradient descent illuminates how gradient vectors, pointing at directions of steepest descent, can be improved by considering the local curvature of loss landscapes. We extend this perspective and show that to fully solve the problem illuminated by natural gradients in neural networks, one must recognise that correlations in the data at any linear transformation, including node responses at every layer of a neural network, cause a non-or"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.10780","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.10780/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.10780","created_at":"2026-07-05T11:58:19.666997+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.10780v4","created_at":"2026-07-05T11:58:19.666997+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.10780","created_at":"2026-07-05T11:58:19.666997+00:00"},{"alias_kind":"pith_short_12","alias_value":"H3QG4QEGD6HV","created_at":"2026-07-05T11:58:19.666997+00:00"},{"alias_kind":"pith_short_16","alias_value":"H3QG4QEGD6HVSURF","created_at":"2026-07-05T11:58:19.666997+00:00"},{"alias_kind":"pith_short_8","alias_value":"H3QG4QEG","created_at":"2026-07-05T11:58:19.666997+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.14336","citing_title":"Mistake gating leads to energy and memory efficient continual learning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI","json":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI.json","graph_json":"https://pith.science/api/pith-number/H3QG4QEGD6HVSURFE6YAEB2RDI/graph.json","events_json":"https://pith.science/api/pith-number/H3QG4QEGD6HVSURFE6YAEB2RDI/events.json","paper":"https://pith.science/paper/H3QG4QEG"},"agent_actions":{"view_html":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI","download_json":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI.json","view_paper":"https://pith.science/paper/H3QG4QEG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.10780&json=true","fetch_graph":"https://pith.science/api/pith-number/H3QG4QEGD6HVSURFE6YAEB2RDI/graph.json","fetch_events":"https://pith.science/api/pith-number/H3QG4QEGD6HVSURFE6YAEB2RDI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI/action/storage_attestation","attest_author":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI/action/author_attestation","sign_citation":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI/action/citation_signature","submit_replication":"https://pith.science/pith/H3QG4QEGD6HVSURFE6YAEB2RDI/action/replication_record"}},"created_at":"2026-07-05T11:58:19.666997+00:00","updated_at":"2026-07-05T11:58:19.666997+00:00"}