{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:V4J6HRH7MEHCFGWVCXTMP4I4RA","short_pith_number":"pith:V4J6HRH7","schema_version":"1.0","canonical_sha256":"af13e3c4ff610e229ad515e6c7f11c881f096e8eb0211c28a3b6ac2d91a435ab","source":{"kind":"arxiv","id":"2108.06325","version":3},"attestation_state":"computed","paper":{"title":"Continual Backprop: Stochastic Gradient Descent with Persistent Randomness","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"A. Rupam Mahmood, Richard S. Sutton, Shibhansh Dohare","submitted_at":"2021-08-13T17:33:47Z","abstract_excerpt":"The Backprop algorithm for learning in neural networks utilizes two mechanisms: first, stochastic gradient descent and second, initialization with small random weights, where the latter is essential to the effectiveness of the former. We show that in continual learning setups, Backprop performs well initially, but over time its performance degrades. Stochastic gradient descent alone is insufficient to learn continually; the initial randomness enables only initial learning but not continual learning. To the best of our knowledge, ours is the first result showing this degradation in Backprop's a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.06325","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2021-08-13T17:33:47Z","cross_cats_sorted":[],"title_canon_sha256":"9907e4fd0664e808627c0cec4fe3384480877baab564c245896bdfbb97f6848b","abstract_canon_sha256":"703190e261ae910b7d6a1887bc14b8e3efc1ea3671da858f16bff3cc81e8fc29"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:20:29.064391Z","signature_b64":"hqmYqvWRk1Cbh1xVnKmLB6sy2fvzjZU/0rZZbAC6vMuA+07+IUKkkoD95b0Gv0yUc1vBtzR1diKEcaQM9JxaAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af13e3c4ff610e229ad515e6c7f11c881f096e8eb0211c28a3b6ac2d91a435ab","last_reissued_at":"2026-07-05T04:20:29.063924Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:20:29.063924Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Continual Backprop: Stochastic Gradient Descent with Persistent Randomness","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"A. Rupam Mahmood, Richard S. Sutton, Shibhansh Dohare","submitted_at":"2021-08-13T17:33:47Z","abstract_excerpt":"The Backprop algorithm for learning in neural networks utilizes two mechanisms: first, stochastic gradient descent and second, initialization with small random weights, where the latter is essential to the effectiveness of the former. We show that in continual learning setups, Backprop performs well initially, but over time its performance degrades. Stochastic gradient descent alone is insufficient to learn continually; the initial randomness enables only initial learning but not continual learning. To the best of our knowledge, ours is the first result showing this degradation in Backprop's a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.06325","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.06325/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.06325","created_at":"2026-07-05T04:20:29.063983+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.06325v3","created_at":"2026-07-05T04:20:29.063983+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.06325","created_at":"2026-07-05T04:20:29.063983+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4J6HRH7MEHC","created_at":"2026-07-05T04:20:29.063983+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4J6HRH7MEHCFGWV","created_at":"2026-07-05T04:20:29.063983+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4J6HRH7","created_at":"2026-07-05T04:20:29.063983+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25335","citing_title":"Stagnant Neuron: Towards Understanding the Plasticity Loss in Multi-Agent Reinforcement Learning Value Factorization Methods","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00304","citing_title":"Barriers for Learning in an Evolving World: Mathematical Understanding of Loss of Plasticity","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16318","citing_title":"Investigating Action Encodings in Recurrent Neural Networks in Reinforcement Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06834","citing_title":"Attribution-Based Neuron Utility for Plasticity Restoration in Deep Networks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18857","citing_title":"Task Switching Without Forgetting via Proximal Decoupling","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15414","citing_title":"Beyond Single-Model Optimization: Preserving Plasticity in Continual Reinforcement Learning","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA","json":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA.json","graph_json":"https://pith.science/api/pith-number/V4J6HRH7MEHCFGWVCXTMP4I4RA/graph.json","events_json":"https://pith.science/api/pith-number/V4J6HRH7MEHCFGWVCXTMP4I4RA/events.json","paper":"https://pith.science/paper/V4J6HRH7"},"agent_actions":{"view_html":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA","download_json":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA.json","view_paper":"https://pith.science/paper/V4J6HRH7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.06325&json=true","fetch_graph":"https://pith.science/api/pith-number/V4J6HRH7MEHCFGWVCXTMP4I4RA/graph.json","fetch_events":"https://pith.science/api/pith-number/V4J6HRH7MEHCFGWVCXTMP4I4RA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA/action/storage_attestation","attest_author":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA/action/author_attestation","sign_citation":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA/action/citation_signature","submit_replication":"https://pith.science/pith/V4J6HRH7MEHCFGWVCXTMP4I4RA/action/replication_record"}},"created_at":"2026-07-05T04:20:29.063983+00:00","updated_at":"2026-07-05T04:20:29.063983+00:00"}