{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JTF3E5SR25HOKC4VQKHYBA6D6U","short_pith_number":"pith:JTF3E5SR","schema_version":"1.0","canonical_sha256":"4ccbb27651d74ee50b95828f8083c3f51dd6ee8cbc2612ebc6509e18dcef1f9c","source":{"kind":"arxiv","id":"2403.06398","version":3},"attestation_state":"computed","paper":{"title":"On the Diminishing Returns of Width for Continual Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Etash Guha, Vihan Lakshman","submitted_at":"2024-03-11T03:19:45Z","abstract_excerpt":"While deep neural networks have demonstrated groundbreaking performance in various settings, these models often suffer from \\emph{catastrophic forgetting} when trained on new tasks in sequence. Several works have empirically demonstrated that increasing the width of a neural network leads to a decrease in catastrophic forgetting but have yet to characterize the exact relationship between width and continual learning. We design one of the first frameworks to analyze Continual Learning Theory and prove that width is directly related to forgetting in Feed-Forward Networks (FFN). Specifically, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.06398","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-11T03:19:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3335851f5ce42021dcb4d4ca25b4bfebe10884b5f84ec94dfd874f493d3205f1","abstract_canon_sha256":"eeb5c1b112a485648661887c24fa54c1be21551196321683b169b7b663cd4c62"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:15.198524Z","signature_b64":"niL6JSG2Re0amu+FhAOyfXjup4r6u0FPpg3TpZUVsrA5Voyi9g9TM+W5Ct1fgb+pUj1ruyLXVAXo7OvCEiWOCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ccbb27651d74ee50b95828f8083c3f51dd6ee8cbc2612ebc6509e18dcef1f9c","last_reissued_at":"2026-07-05T08:34:15.198129Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:15.198129Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Diminishing Returns of Width for Continual Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Etash Guha, Vihan Lakshman","submitted_at":"2024-03-11T03:19:45Z","abstract_excerpt":"While deep neural networks have demonstrated groundbreaking performance in various settings, these models often suffer from \\emph{catastrophic forgetting} when trained on new tasks in sequence. Several works have empirically demonstrated that increasing the width of a neural network leads to a decrease in catastrophic forgetting but have yet to characterize the exact relationship between width and continual learning. We design one of the first frameworks to analyze Continual Learning Theory and prove that width is directly related to forgetting in Feed-Forward Networks (FFN). Specifically, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.06398","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.06398/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.06398","created_at":"2026-07-05T08:34:15.198185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.06398v3","created_at":"2026-07-05T08:34:15.198185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.06398","created_at":"2026-07-05T08:34:15.198185+00:00"},{"alias_kind":"pith_short_12","alias_value":"JTF3E5SR25HO","created_at":"2026-07-05T08:34:15.198185+00:00"},{"alias_kind":"pith_short_16","alias_value":"JTF3E5SR25HOKC4V","created_at":"2026-07-05T08:34:15.198185+00:00"},{"alias_kind":"pith_short_8","alias_value":"JTF3E5SR","created_at":"2026-07-05T08:34:15.198185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20970","citing_title":"Measuring Representational Shifts in Continual Learning: A Linear Transformation Perspective","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U","json":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U.json","graph_json":"https://pith.science/api/pith-number/JTF3E5SR25HOKC4VQKHYBA6D6U/graph.json","events_json":"https://pith.science/api/pith-number/JTF3E5SR25HOKC4VQKHYBA6D6U/events.json","paper":"https://pith.science/paper/JTF3E5SR"},"agent_actions":{"view_html":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U","download_json":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U.json","view_paper":"https://pith.science/paper/JTF3E5SR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.06398&json=true","fetch_graph":"https://pith.science/api/pith-number/JTF3E5SR25HOKC4VQKHYBA6D6U/graph.json","fetch_events":"https://pith.science/api/pith-number/JTF3E5SR25HOKC4VQKHYBA6D6U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U/action/storage_attestation","attest_author":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U/action/author_attestation","sign_citation":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U/action/citation_signature","submit_replication":"https://pith.science/pith/JTF3E5SR25HOKC4VQKHYBA6D6U/action/replication_record"}},"created_at":"2026-07-05T08:34:15.198185+00:00","updated_at":"2026-07-05T08:34:15.198185+00:00"}