{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WKDAICSOW4NSFXUXKCJHJXRQCZ","short_pith_number":"pith:WKDAICSO","schema_version":"1.0","canonical_sha256":"b286040a4eb71b22de97509274de301640822862d61a653902eb98b2bf12e333","source":{"kind":"arxiv","id":"2205.06226","version":3},"attestation_state":"computed","paper":{"title":"The Mechanism of Prediction Head in Non-contrastive Self-supervised Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Yuanzhi Li, Zixin Wen","submitted_at":"2022-05-12T17:15:53Z","abstract_excerpt":"Recently the surprising discovery of the Bootstrap Your Own Latent (BYOL) method by Grill et al. shows the negative term in contrastive loss can be removed if we add the so-called prediction head to the network. This initiated the research of non-contrastive self-supervised learning. It is mysterious why even when there exist trivial collapsed global optimal solutions, neural networks trained by (stochastic) gradient descent can still learn competitive representations. This phenomenon is a typical example of implicit bias in deep learning and remains little understood.\n  In this work, we prese"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.06226","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-12T17:15:53Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"535fadea650ffd49df0ebff3a3f3292ef2459497e102eff60905d4ab8664e76d","abstract_canon_sha256":"3885490c224569b53e5e979f31f0de23d3c7e91f405fc1f635e82f04abc10302"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:33:10.827186Z","signature_b64":"JiBkQZISlAcpQGX2xUTrTZAcLY5/y7bHMMkbKZgE3H+5I8q63CdJD1SdS0/xi0IE4Khg9r0qoqltOc5Moq2oBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b286040a4eb71b22de97509274de301640822862d61a653902eb98b2bf12e333","last_reissued_at":"2026-07-05T05:33:10.826671Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:33:10.826671Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Mechanism of Prediction Head in Non-contrastive Self-supervised Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Yuanzhi Li, Zixin Wen","submitted_at":"2022-05-12T17:15:53Z","abstract_excerpt":"Recently the surprising discovery of the Bootstrap Your Own Latent (BYOL) method by Grill et al. shows the negative term in contrastive loss can be removed if we add the so-called prediction head to the network. This initiated the research of non-contrastive self-supervised learning. It is mysterious why even when there exist trivial collapsed global optimal solutions, neural networks trained by (stochastic) gradient descent can still learn competitive representations. This phenomenon is a typical example of implicit bias in deep learning and remains little understood.\n  In this work, we prese"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.06226","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.06226/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.06226","created_at":"2026-07-05T05:33:10.826741+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.06226v3","created_at":"2026-07-05T05:33:10.826741+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.06226","created_at":"2026-07-05T05:33:10.826741+00:00"},{"alias_kind":"pith_short_12","alias_value":"WKDAICSOW4NS","created_at":"2026-07-05T05:33:10.826741+00:00"},{"alias_kind":"pith_short_16","alias_value":"WKDAICSOW4NSFXUX","created_at":"2026-07-05T05:33:10.826741+00:00"},{"alias_kind":"pith_short_8","alias_value":"WKDAICSO","created_at":"2026-07-05T05:33:10.826741+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2211.16327","citing_title":"On the Power of Foundation Models","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ","json":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ.json","graph_json":"https://pith.science/api/pith-number/WKDAICSOW4NSFXUXKCJHJXRQCZ/graph.json","events_json":"https://pith.science/api/pith-number/WKDAICSOW4NSFXUXKCJHJXRQCZ/events.json","paper":"https://pith.science/paper/WKDAICSO"},"agent_actions":{"view_html":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ","download_json":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ.json","view_paper":"https://pith.science/paper/WKDAICSO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.06226&json=true","fetch_graph":"https://pith.science/api/pith-number/WKDAICSOW4NSFXUXKCJHJXRQCZ/graph.json","fetch_events":"https://pith.science/api/pith-number/WKDAICSOW4NSFXUXKCJHJXRQCZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ/action/storage_attestation","attest_author":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ/action/author_attestation","sign_citation":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ/action/citation_signature","submit_replication":"https://pith.science/pith/WKDAICSOW4NSFXUXKCJHJXRQCZ/action/replication_record"}},"created_at":"2026-07-05T05:33:10.826741+00:00","updated_at":"2026-07-05T05:33:10.826741+00:00"}