{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GZ5V7KU5JT7EFFEPXKDTFZORJJ","short_pith_number":"pith:GZ5V7KU5","schema_version":"1.0","canonical_sha256":"367b5faa9d4cfe42948fba8732e5d14a42f4e4c2001d480347115a6180ba725f","source":{"kind":"arxiv","id":"2507.17912","version":2},"attestation_state":"computed","paper":{"title":"SETOL: A Semi-Empirical Theory of (Deep) Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.stat-mech"],"primary_cat":"cs.LG","authors_text":"Charles H Martin, Christopher Hinrichs","submitted_at":"2025-07-23T20:22:20Z","abstract_excerpt":"We present a SemiEmpirical Theory of Learning (SETOL) that explains the remarkable performance of State-Of-The-Art (SOTA) Neural Networks (NNs). We provide a formal explanation of the origin of the fundamental quantities in the phenomenological theory of Heavy-Tailed Self-Regularization (HTSR): the heavy-tailed power-law layer quality metrics, alpha and alpha-hat. In prior work, these metrics have been shown to predict trends in the test accuracies of pretrained SOTA NN models, importantly, without needing access to either testing or training data. Our SETOL uses techniques from statistical me"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.17912","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-23T20:22:20Z","cross_cats_sorted":["cond-mat.stat-mech"],"title_canon_sha256":"71bd8c73db9ccfb90c313c764ddf15bf3e1158e4cdc2e89bcd3a3ce56b5bb58a","abstract_canon_sha256":"d470667ecb14c5ddf35f1d6288268697062879b8d49235858e657f3374be5173"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:10.956237Z","signature_b64":"c+TJI5B3poVg+QQQbjnpJ9OrdCy7hJ1XMC2934NDb17S19m4SILrR1tgYdNQBbKeIEQfj+uaDnv/Hkxhdi5QDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"367b5faa9d4cfe42948fba8732e5d14a42f4e4c2001d480347115a6180ba725f","last_reissued_at":"2026-07-05T11:44:10.955712Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:10.955712Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SETOL: A Semi-Empirical Theory of (Deep) Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.stat-mech"],"primary_cat":"cs.LG","authors_text":"Charles H Martin, Christopher Hinrichs","submitted_at":"2025-07-23T20:22:20Z","abstract_excerpt":"We present a SemiEmpirical Theory of Learning (SETOL) that explains the remarkable performance of State-Of-The-Art (SOTA) Neural Networks (NNs). We provide a formal explanation of the origin of the fundamental quantities in the phenomenological theory of Heavy-Tailed Self-Regularization (HTSR): the heavy-tailed power-law layer quality metrics, alpha and alpha-hat. In prior work, these metrics have been shown to predict trends in the test accuracies of pretrained SOTA NN models, importantly, without needing access to either testing or training data. Our SETOL uses techniques from statistical me"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.17912","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.17912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.17912","created_at":"2026-07-05T11:44:10.955772+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.17912v2","created_at":"2026-07-05T11:44:10.955772+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.17912","created_at":"2026-07-05T11:44:10.955772+00:00"},{"alias_kind":"pith_short_12","alias_value":"GZ5V7KU5JT7E","created_at":"2026-07-05T11:44:10.955772+00:00"},{"alias_kind":"pith_short_16","alias_value":"GZ5V7KU5JT7EFFEP","created_at":"2026-07-05T11:44:10.955772+00:00"},{"alias_kind":"pith_short_8","alias_value":"GZ5V7KU5","created_at":"2026-07-05T11:44:10.955772+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19268","citing_title":"Patnaik-Pearson intrinsic dimension for internal representations of neural networks","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19268","citing_title":"Patnaik-Pearson intrinsic dimension for internal representations of neural networks","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01521","citing_title":"Fast Generalization after Interpolation via Critically Damped Momentum Optimization","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15551","citing_title":"Characterizing Learning in Deep Neural Networks using Tractable Algorithmic Complexity Analysis","ref_index":298,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12394","citing_title":"Detecting overfitting in Neural Networks during long-horizon grokking using Random Matrix Theory","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12394","citing_title":"Detecting overfitting in Neural Networks during long-horizon grokking using Random Matrix Theory","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ","json":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ.json","graph_json":"https://pith.science/api/pith-number/GZ5V7KU5JT7EFFEPXKDTFZORJJ/graph.json","events_json":"https://pith.science/api/pith-number/GZ5V7KU5JT7EFFEPXKDTFZORJJ/events.json","paper":"https://pith.science/paper/GZ5V7KU5"},"agent_actions":{"view_html":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ","download_json":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ.json","view_paper":"https://pith.science/paper/GZ5V7KU5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.17912&json=true","fetch_graph":"https://pith.science/api/pith-number/GZ5V7KU5JT7EFFEPXKDTFZORJJ/graph.json","fetch_events":"https://pith.science/api/pith-number/GZ5V7KU5JT7EFFEPXKDTFZORJJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ/action/storage_attestation","attest_author":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ/action/author_attestation","sign_citation":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ/action/citation_signature","submit_replication":"https://pith.science/pith/GZ5V7KU5JT7EFFEPXKDTFZORJJ/action/replication_record"}},"created_at":"2026-07-05T11:44:10.955772+00:00","updated_at":"2026-07-05T11:44:10.955772+00:00"}