{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:FHJAPJFADUIXXXYREOBHAINVNL","short_pith_number":"pith:FHJAPJFA","schema_version":"1.0","canonical_sha256":"29d207a4a01d117bdf1123827021b56acbb9dd7aa8d798bd01435753e9ee5085","source":{"kind":"arxiv","id":"2206.13378","version":2},"attestation_state":"computed","paper":{"title":"Guillotine Regularization: Why removing layers is needed to improve generalization in Self-Supervised Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adrien Bardes, Florian Bordes, Pascal Vincent, Quentin Garrido, Randall Balestriero","submitted_at":"2022-06-27T15:37:54Z","abstract_excerpt":"One unexpected technique that emerged in recent years consists in training a Deep Network (DN) with a Self-Supervised Learning (SSL) method, and using this network on downstream tasks but with its last few projector layers entirely removed. This trick of throwing away the projector is actually critical for SSL methods to display competitive performances on ImageNet for which more than 30 percentage points can be gained that way. This is a little vexing, as one would hope that the network layer at which invariance is explicitly enforced by the SSL criterion during training (the last projector l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.13378","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-27T15:37:54Z","cross_cats_sorted":[],"title_canon_sha256":"a311ac205a771f6c8b8d8e158dbc96d3485eab77275e106f21f539f576c79053","abstract_canon_sha256":"a9773eb4a858cd5184ef8f07ea1bb74996b9ba693b1ec2537ff1c32deab5a3c0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:18:59.230047Z","signature_b64":"cefAHCTyYWbahN4C6EawnTrG5PnmvbKRlubVFXcDU5smSJ9IIxcP5SE2PXC/MmaBN/SAfRKrHXOYuU3ajNf8AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29d207a4a01d117bdf1123827021b56acbb9dd7aa8d798bd01435753e9ee5085","last_reissued_at":"2026-07-05T06:18:59.229649Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:18:59.229649Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Guillotine Regularization: Why removing layers is needed to improve generalization in Self-Supervised Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adrien Bardes, Florian Bordes, Pascal Vincent, Quentin Garrido, Randall Balestriero","submitted_at":"2022-06-27T15:37:54Z","abstract_excerpt":"One unexpected technique that emerged in recent years consists in training a Deep Network (DN) with a Self-Supervised Learning (SSL) method, and using this network on downstream tasks but with its last few projector layers entirely removed. This trick of throwing away the projector is actually critical for SSL methods to display competitive performances on ImageNet for which more than 30 percentage points can be gained that way. This is a little vexing, as one would hope that the network layer at which invariance is explicitly enforced by the SSL criterion during training (the last projector l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.13378","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.13378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.13378","created_at":"2026-07-05T06:18:59.229706+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.13378v2","created_at":"2026-07-05T06:18:59.229706+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.13378","created_at":"2026-07-05T06:18:59.229706+00:00"},{"alias_kind":"pith_short_12","alias_value":"FHJAPJFADUIX","created_at":"2026-07-05T06:18:59.229706+00:00"},{"alias_kind":"pith_short_16","alias_value":"FHJAPJFADUIXXXYR","created_at":"2026-07-05T06:18:59.229706+00:00"},{"alias_kind":"pith_short_8","alias_value":"FHJAPJFA","created_at":"2026-07-05T06:18:59.229706+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00958","citing_title":"LeNEPA: No-Augmentation Next-Latent Prediction for Time-Series Representation Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21590","citing_title":"Radial Basis Function Networks as Projection Heads in Self-Supervised Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13181","citing_title":"Perception Encoder: The best visual embeddings are not at the output of the network","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08471","citing_title":"Revisiting Feature Prediction for Learning Visual Representations from Video","ref_index":139,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL","json":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL.json","graph_json":"https://pith.science/api/pith-number/FHJAPJFADUIXXXYREOBHAINVNL/graph.json","events_json":"https://pith.science/api/pith-number/FHJAPJFADUIXXXYREOBHAINVNL/events.json","paper":"https://pith.science/paper/FHJAPJFA"},"agent_actions":{"view_html":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL","download_json":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL.json","view_paper":"https://pith.science/paper/FHJAPJFA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.13378&json=true","fetch_graph":"https://pith.science/api/pith-number/FHJAPJFADUIXXXYREOBHAINVNL/graph.json","fetch_events":"https://pith.science/api/pith-number/FHJAPJFADUIXXXYREOBHAINVNL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL/action/storage_attestation","attest_author":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL/action/author_attestation","sign_citation":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL/action/citation_signature","submit_replication":"https://pith.science/pith/FHJAPJFADUIXXXYREOBHAINVNL/action/replication_record"}},"created_at":"2026-07-05T06:18:59.229706+00:00","updated_at":"2026-07-05T06:18:59.229706+00:00"}