{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:TCYPAKRFIOKJQTX4VJ4IT556C3","short_pith_number":"pith:TCYPAKRF","schema_version":"1.0","canonical_sha256":"98b0f02a254394984efcaa7889f7be16cec9505d8257cddf29324c397a25cbf9","source":{"kind":"arxiv","id":"2011.10566","version":1},"attestation_state":"computed","paper":{"title":"Exploring Simple Siamese Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Kaiming He, Xinlei Chen","submitted_at":"2020-11-20T18:59:33Z","abstract_excerpt":"Siamese networks have become a common structure in various recent models for unsupervised visual representation learning. These models maximize the similarity between two augmentations of one image, subject to certain conditions for avoiding collapsing solutions. In this paper, we report surprising empirical results that simple Siamese networks can learn meaningful representations even using none of the following: (i) negative sample pairs, (ii) large batches, (iii) momentum encoders. Our experiments show that collapsing solutions do exist for the loss and structure, but a stop-gradient operat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.10566","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-11-20T18:59:33Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9133673eccd81baf49ec89514aaf591c9796e476a8a0436f0da55bbbfd14d0c5","abstract_canon_sha256":"9978cb28fd61c25fd5ecba303dcd161e7248fd79a46689299394a3cd3e6f7ff3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:53:16.168250Z","signature_b64":"GljAuEORvmVxsTFcf0XFgaldZS+Rrcra4LCWFKnCM6HgjIbErAgAIke4njanwF0I5viyDvokZl4v9mzjSvz0BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98b0f02a254394984efcaa7889f7be16cec9505d8257cddf29324c397a25cbf9","last_reissued_at":"2026-07-05T01:53:16.167801Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:53:16.167801Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Simple Siamese Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Kaiming He, Xinlei Chen","submitted_at":"2020-11-20T18:59:33Z","abstract_excerpt":"Siamese networks have become a common structure in various recent models for unsupervised visual representation learning. These models maximize the similarity between two augmentations of one image, subject to certain conditions for avoiding collapsing solutions. In this paper, we report surprising empirical results that simple Siamese networks can learn meaningful representations even using none of the following: (i) negative sample pairs, (ii) large batches, (iii) momentum encoders. Our experiments show that collapsing solutions do exist for the loss and structure, but a stop-gradient operat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.10566","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.10566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.10566","created_at":"2026-07-05T01:53:16.167859+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.10566v1","created_at":"2026-07-05T01:53:16.167859+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.10566","created_at":"2026-07-05T01:53:16.167859+00:00"},{"alias_kind":"pith_short_12","alias_value":"TCYPAKRFIOKJ","created_at":"2026-07-05T01:53:16.167859+00:00"},{"alias_kind":"pith_short_16","alias_value":"TCYPAKRFIOKJQTX4","created_at":"2026-07-05T01:53:16.167859+00:00"},{"alias_kind":"pith_short_8","alias_value":"TCYPAKRF","created_at":"2026-07-05T01:53:16.167859+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26091","citing_title":"On-Policy Self-Distillation with Sampled Demonstrations Reduces Output Diversity","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23611","citing_title":"Data Selection Through Iterative Self-Filtering for Vision-Language Settings","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01145","citing_title":"Hierarchical Self-Supervised Representation Learning Framework for Multivariate Time Series Grounded in ECG Analysis","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01443","citing_title":"UR-JEPA: Uniform Rectifiability as a Regularizer for Joint-Embedding Predictive Architectures","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25242","citing_title":"C3P: Contrastive promoter-protein pretraining yields representations capturing bacterial gene regulation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25313","citing_title":"UWM-JEPA: Predictive World Models That Imagine in Belief Space","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28990","citing_title":"Learning Robust and Task-Invariant Functional Representation from fMRI through Siamese Self-Supervised Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11190","citing_title":"When to Align, When to Predict: A Phase Diagram for Multimodal Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26059","citing_title":"A welding penetration prediction model for laser welding process based on self-supervised learning using physics-informed neural networks","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2410.08559","citing_title":"Learning General Representation of 12-Lead Electrocardiogram with a Joint-Embedding Predictive Architecture","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15394","citing_title":"Representation Without Reward: A JEPA Audit for LLM Fine-Tuning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2104.14294","citing_title":"Emerging Properties in Self-Supervised Vision Transformers","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2106.08254","citing_title":"BEiT: BERT Pre-Training of Image Transformers","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08471","citing_title":"Revisiting Feature Prediction for Learning Visual Representations from Video","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22631","citing_title":"Identifying and typifying demographic unfairness in phoneme-level embeddings of self-supervised speech recognition models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06266","citing_title":"ZScribbleSeg: A comprehensive segmentation framework with modeling of efficient annotation and maximization of scribble supervision","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3","json":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3.json","graph_json":"https://pith.science/api/pith-number/TCYPAKRFIOKJQTX4VJ4IT556C3/graph.json","events_json":"https://pith.science/api/pith-number/TCYPAKRFIOKJQTX4VJ4IT556C3/events.json","paper":"https://pith.science/paper/TCYPAKRF"},"agent_actions":{"view_html":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3","download_json":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3.json","view_paper":"https://pith.science/paper/TCYPAKRF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.10566&json=true","fetch_graph":"https://pith.science/api/pith-number/TCYPAKRFIOKJQTX4VJ4IT556C3/graph.json","fetch_events":"https://pith.science/api/pith-number/TCYPAKRFIOKJQTX4VJ4IT556C3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3/action/storage_attestation","attest_author":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3/action/author_attestation","sign_citation":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3/action/citation_signature","submit_replication":"https://pith.science/pith/TCYPAKRFIOKJQTX4VJ4IT556C3/action/replication_record"}},"created_at":"2026-07-05T01:53:16.167859+00:00","updated_at":"2026-07-05T01:53:16.167859+00:00"}