{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7BF45JC4REJO2LRW3FGO4YSAXT","short_pith_number":"pith:7BF45JC4","schema_version":"1.0","canonical_sha256":"f84bcea45c8912ed2e36d94cee6240bcc5d0ba98969842ba872bec066a3f5941","source":{"kind":"arxiv","id":"2402.05271","version":4},"attestation_state":"computed","paper":{"title":"Feature learning as alignment: a structural property of gradient descent in non-linear neural networks","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"stat.ML","authors_text":"Atish Agarwala, Daniel Beaglehole, Ioannis Mitliagkas","submitted_at":"2024-02-07T21:31:53Z","abstract_excerpt":"Understanding the mechanisms through which neural networks extract statistics from input-label pairs through feature learning is one of the most important unsolved problems in supervised learning. Prior works demonstrated that the gram matrices of the weights (the neural feature matrices, NFM) and the average gradient outer products (AGOP) become correlated during training, in a statement known as the neural feature ansatz (NFA). Through the NFA, the authors introduce mapping with the AGOP as a general mechanism for neural feature learning. However, these works do not provide a theoretical exp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05271","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"stat.ML","submitted_at":"2024-02-07T21:31:53Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d0cc469618859dc5d6e85685516aeb56058a9b9ea41e14b378b72f5a027c26ee","abstract_canon_sha256":"bd639fa89d4e23ebc9a2816a4f326c5ec82de892b0e87d6af7295d3724452253"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:36:38.566714Z","signature_b64":"39DOrO5gQSU4QCPtzvMn48wb0rVi3/w1psiAKozoPdElpIpGYln2LrOiajMZaAuRtHHbz/oDK7Ag53gp3uzPCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f84bcea45c8912ed2e36d94cee6240bcc5d0ba98969842ba872bec066a3f5941","last_reissued_at":"2026-07-05T09:36:38.566047Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:36:38.566047Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Feature learning as alignment: a structural property of gradient descent in non-linear neural networks","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"stat.ML","authors_text":"Atish Agarwala, Daniel Beaglehole, Ioannis Mitliagkas","submitted_at":"2024-02-07T21:31:53Z","abstract_excerpt":"Understanding the mechanisms through which neural networks extract statistics from input-label pairs through feature learning is one of the most important unsolved problems in supervised learning. Prior works demonstrated that the gram matrices of the weights (the neural feature matrices, NFM) and the average gradient outer products (AGOP) become correlated during training, in a statement known as the neural feature ansatz (NFA). Through the NFA, the authors introduce mapping with the AGOP as a general mechanism for neural feature learning. However, these works do not provide a theoretical exp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05271","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05271/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05271","created_at":"2026-07-05T09:36:38.566115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05271v4","created_at":"2026-07-05T09:36:38.566115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05271","created_at":"2026-07-05T09:36:38.566115+00:00"},{"alias_kind":"pith_short_12","alias_value":"7BF45JC4REJO","created_at":"2026-07-05T09:36:38.566115+00:00"},{"alias_kind":"pith_short_16","alias_value":"7BF45JC4REJO2LRW","created_at":"2026-07-05T09:36:38.566115+00:00"},{"alias_kind":"pith_short_8","alias_value":"7BF45JC4","created_at":"2026-07-05T09:36:38.566115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19367","citing_title":"Weibull Weight-Scale Parameter Evolution under AdamW Training Dynamics","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04971","citing_title":"Why Geometric Continuity Emerges in Deep Neural Networks: Residual Connections and Rotational Symmetry Breaking","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT","json":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT.json","graph_json":"https://pith.science/api/pith-number/7BF45JC4REJO2LRW3FGO4YSAXT/graph.json","events_json":"https://pith.science/api/pith-number/7BF45JC4REJO2LRW3FGO4YSAXT/events.json","paper":"https://pith.science/paper/7BF45JC4"},"agent_actions":{"view_html":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT","download_json":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT.json","view_paper":"https://pith.science/paper/7BF45JC4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05271&json=true","fetch_graph":"https://pith.science/api/pith-number/7BF45JC4REJO2LRW3FGO4YSAXT/graph.json","fetch_events":"https://pith.science/api/pith-number/7BF45JC4REJO2LRW3FGO4YSAXT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT/action/storage_attestation","attest_author":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT/action/author_attestation","sign_citation":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT/action/citation_signature","submit_replication":"https://pith.science/pith/7BF45JC4REJO2LRW3FGO4YSAXT/action/replication_record"}},"created_at":"2026-07-05T09:36:38.566115+00:00","updated_at":"2026-07-05T09:36:38.566115+00:00"}