{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:M53PNWTGQFMF3KX26TKJ6QGSPZ","short_pith_number":"pith:M53PNWTG","schema_version":"1.0","canonical_sha256":"6776f6da6681585daafaf4d49f40d27e7f3afa93689bbad05fbd4e7b181127e2","source":{"kind":"arxiv","id":"2107.05802","version":2},"attestation_state":"computed","paper":{"title":"How many degrees of freedom do we need to train deep networks: a loss landscape perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Brett W. Larsen, Nic Becker, Stanislav Fort, Surya Ganguli","submitted_at":"2021-07-13T01:29:24Z","abstract_excerpt":"A variety of recent works, spanning pruning, lottery tickets, and training within random subspaces, have shown that deep neural networks can be trained using far fewer degrees of freedom than the total number of parameters. We analyze this phenomenon for random subspaces by first examining the success probability of hitting a training loss sub-level set when training within a random subspace of a given training dimensionality. We find a sharp phase transition in the success probability from $0$ to $1$ as the training dimension surpasses a threshold. This threshold training dimension increases "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.05802","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-07-13T01:29:24Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"6bfa69cd1065fe06036b84438f389e409f7079b095d6f6f50fea77d70b872fe4","abstract_canon_sha256":"7ebd78a537c482c5e0e338986d1fb6ff08261f21d526f67945a573aa2ba33249"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:53:48.387930Z","signature_b64":"NfLhhJDwh73mzKN5dKTI41oaXt+mXZshxrtxGDWuQRXP5KQ58p4IBPGhxUR7JQgWcMayQCyPfdOcFf4dUPzlBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6776f6da6681585daafaf4d49f40d27e7f3afa93689bbad05fbd4e7b181127e2","last_reissued_at":"2026-07-05T03:53:48.387489Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:53:48.387489Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How many degrees of freedom do we need to train deep networks: a loss landscape perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Brett W. Larsen, Nic Becker, Stanislav Fort, Surya Ganguli","submitted_at":"2021-07-13T01:29:24Z","abstract_excerpt":"A variety of recent works, spanning pruning, lottery tickets, and training within random subspaces, have shown that deep neural networks can be trained using far fewer degrees of freedom than the total number of parameters. We analyze this phenomenon for random subspaces by first examining the success probability of hitting a training loss sub-level set when training within a random subspace of a given training dimensionality. We find a sharp phase transition in the success probability from $0$ to $1$ as the training dimension surpasses a threshold. This threshold training dimension increases "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.05802","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.05802/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.05802","created_at":"2026-07-05T03:53:48.387546+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.05802v2","created_at":"2026-07-05T03:53:48.387546+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.05802","created_at":"2026-07-05T03:53:48.387546+00:00"},{"alias_kind":"pith_short_12","alias_value":"M53PNWTGQFMF","created_at":"2026-07-05T03:53:48.387546+00:00"},{"alias_kind":"pith_short_16","alias_value":"M53PNWTGQFMF3KX2","created_at":"2026-07-05T03:53:48.387546+00:00"},{"alias_kind":"pith_short_8","alias_value":"M53PNWTG","created_at":"2026-07-05T03:53:48.387546+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.21226","citing_title":"Why Do More Experts Fail? A Theoretical Analysis of Model Merging","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ","json":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ.json","graph_json":"https://pith.science/api/pith-number/M53PNWTGQFMF3KX26TKJ6QGSPZ/graph.json","events_json":"https://pith.science/api/pith-number/M53PNWTGQFMF3KX26TKJ6QGSPZ/events.json","paper":"https://pith.science/paper/M53PNWTG"},"agent_actions":{"view_html":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ","download_json":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ.json","view_paper":"https://pith.science/paper/M53PNWTG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.05802&json=true","fetch_graph":"https://pith.science/api/pith-number/M53PNWTGQFMF3KX26TKJ6QGSPZ/graph.json","fetch_events":"https://pith.science/api/pith-number/M53PNWTGQFMF3KX26TKJ6QGSPZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ/action/storage_attestation","attest_author":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ/action/author_attestation","sign_citation":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ/action/citation_signature","submit_replication":"https://pith.science/pith/M53PNWTGQFMF3KX26TKJ6QGSPZ/action/replication_record"}},"created_at":"2026-07-05T03:53:48.387546+00:00","updated_at":"2026-07-05T03:53:48.387546+00:00"}