{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DVLYFFK4BXGHFHP7N7UDHYFSRX","short_pith_number":"pith:DVLYFFK4","schema_version":"1.0","canonical_sha256":"1d5782955c0dcc729dff6fe833e0b28dfdfe854554d00a22afcba773ca0f1416","source":{"kind":"arxiv","id":"2211.08403","version":3},"attestation_state":"computed","paper":{"title":"REPAIR: REnormalizing Permuted Activations for Interpolation Repair","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Behnam Neyshabur, Hanie Sedghi, Keller Jordan, Olga Saukh, Rahim Entezari","submitted_at":"2022-11-15T18:45:26Z","abstract_excerpt":"In this paper we look into the conjecture of Entezari et al. (2021) which states that if the permutation invariance of neural networks is taken into account, then there is likely no loss barrier to the linear interpolation between SGD solutions. First, we observe that neuron alignment methods alone are insufficient to establish low-barrier linear connectivity between SGD solutions due to a phenomenon we call variance collapse: interpolated deep networks suffer a collapse in the variance of their activations, causing poor performance. Next, we propose REPAIR (REnormalizing Permuted Activations "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.08403","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2022-11-15T18:45:26Z","cross_cats_sorted":["cs.AI","cs.CV","stat.ML"],"title_canon_sha256":"6920d3f60e5646e20489e5c40f28817e1d2debd8a15cf36d78365885ba758be9","abstract_canon_sha256":"37924f21efb56997458b6d1873a8bf2bebc11a03c3a6b5e4bd1ce04648f64439"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:54:08.584336Z","signature_b64":"KCNsodbc9RJKfHXCaUD0Be7mdXpp+3wVYn/T/hIO+e2F2buxynET6C7UMCY/9859KPJKWLjE0B/wFbKr7M2FAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d5782955c0dcc729dff6fe833e0b28dfdfe854554d00a22afcba773ca0f1416","last_reissued_at":"2026-07-05T06:54:08.583981Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:54:08.583981Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"REPAIR: REnormalizing Permuted Activations for Interpolation Repair","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Behnam Neyshabur, Hanie Sedghi, Keller Jordan, Olga Saukh, Rahim Entezari","submitted_at":"2022-11-15T18:45:26Z","abstract_excerpt":"In this paper we look into the conjecture of Entezari et al. (2021) which states that if the permutation invariance of neural networks is taken into account, then there is likely no loss barrier to the linear interpolation between SGD solutions. First, we observe that neuron alignment methods alone are insufficient to establish low-barrier linear connectivity between SGD solutions due to a phenomenon we call variance collapse: interpolated deep networks suffer a collapse in the variance of their activations, causing poor performance. Next, we propose REPAIR (REnormalizing Permuted Activations "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.08403","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.08403/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.08403","created_at":"2026-07-05T06:54:08.584036+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.08403v3","created_at":"2026-07-05T06:54:08.584036+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.08403","created_at":"2026-07-05T06:54:08.584036+00:00"},{"alias_kind":"pith_short_12","alias_value":"DVLYFFK4BXGH","created_at":"2026-07-05T06:54:08.584036+00:00"},{"alias_kind":"pith_short_16","alias_value":"DVLYFFK4BXGHFHP7","created_at":"2026-07-05T06:54:08.584036+00:00"},{"alias_kind":"pith_short_8","alias_value":"DVLYFFK4","created_at":"2026-07-05T06:54:08.584036+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21638","citing_title":"Toward Open Weight Models Without Risks: Separating Public and Private Capabilities in LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01686","citing_title":"WARP: Weight-Space Analysis for Recovering Training Data Portfolios","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28932","citing_title":"DLR: Zero-Inference-Cost Latent Residuals for Low-Rank Pre-Training","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01674","citing_title":"Can Heterogeneous Language Models Be Fused?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2405.07987","citing_title":"The Platonic Representation Hypothesis","ref_index":259,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX","json":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX.json","graph_json":"https://pith.science/api/pith-number/DVLYFFK4BXGHFHP7N7UDHYFSRX/graph.json","events_json":"https://pith.science/api/pith-number/DVLYFFK4BXGHFHP7N7UDHYFSRX/events.json","paper":"https://pith.science/paper/DVLYFFK4"},"agent_actions":{"view_html":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX","download_json":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX.json","view_paper":"https://pith.science/paper/DVLYFFK4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.08403&json=true","fetch_graph":"https://pith.science/api/pith-number/DVLYFFK4BXGHFHP7N7UDHYFSRX/graph.json","fetch_events":"https://pith.science/api/pith-number/DVLYFFK4BXGHFHP7N7UDHYFSRX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX/action/storage_attestation","attest_author":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX/action/author_attestation","sign_citation":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX/action/citation_signature","submit_replication":"https://pith.science/pith/DVLYFFK4BXGHFHP7N7UDHYFSRX/action/replication_record"}},"created_at":"2026-07-05T06:54:08.584036+00:00","updated_at":"2026-07-05T06:54:08.584036+00:00"}