{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:SE7CS5GLV7Q2R5LFUOIDRLC7AF","short_pith_number":"pith:SE7CS5GL","schema_version":"1.0","canonical_sha256":"913e2974cbafe1a8f565a39038ac5f0166ed552f80f62602c48f1815b8744345","source":{"kind":"arxiv","id":"1911.05722","version":3},"attestation_state":"computed","paper":{"title":"Momentum Contrast for Unsupervised Visual Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoqi Fan, Kaiming He, Ross Girshick, Saining Xie, Yuxin Wu","submitted_at":"2019-11-13T18:53:26Z","abstract_excerpt":"We present Momentum Contrast (MoCo) for unsupervised visual representation learning. From a perspective on contrastive learning as dictionary look-up, we build a dynamic dictionary with a queue and a moving-averaged encoder. This enables building a large and consistent dictionary on-the-fly that facilitates contrastive unsupervised learning. MoCo provides competitive results under the common linear protocol on ImageNet classification. More importantly, the representations learned by MoCo transfer well to downstream tasks. MoCo can outperform its supervised pre-training counterpart in 7 detecti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.05722","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-11-13T18:53:26Z","cross_cats_sorted":[],"title_canon_sha256":"4d981e81e3d1d5470b97869d46dad2bcc46df88719bc31fe4d625eb700e6d17e","abstract_canon_sha256":"a892603ceec03eb438f1af722d67826af82b12b5c9dbca76312acf9c79797af8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:49:52.991121Z","signature_b64":"n+tSuozsoVIPKsrtpLAahQ/Sjn0035sMGTRyHaFBS2+eE0p1ifti5CD9CKz1k/2rz81MocjG5lHhKlwagOGvBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"913e2974cbafe1a8f565a39038ac5f0166ed552f80f62602c48f1815b8744345","last_reissued_at":"2026-07-05T00:49:52.990571Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:49:52.990571Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Momentum Contrast for Unsupervised Visual Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haoqi Fan, Kaiming He, Ross Girshick, Saining Xie, Yuxin Wu","submitted_at":"2019-11-13T18:53:26Z","abstract_excerpt":"We present Momentum Contrast (MoCo) for unsupervised visual representation learning. From a perspective on contrastive learning as dictionary look-up, we build a dynamic dictionary with a queue and a moving-averaged encoder. This enables building a large and consistent dictionary on-the-fly that facilitates contrastive unsupervised learning. MoCo provides competitive results under the common linear protocol on ImageNet classification. More importantly, the representations learned by MoCo transfer well to downstream tasks. MoCo can outperform its supervised pre-training counterpart in 7 detecti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.05722","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.05722/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.05722","created_at":"2026-07-05T00:49:52.990632+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.05722v3","created_at":"2026-07-05T00:49:52.990632+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.05722","created_at":"2026-07-05T00:49:52.990632+00:00"},{"alias_kind":"pith_short_12","alias_value":"SE7CS5GLV7Q2","created_at":"2026-07-05T00:49:52.990632+00:00"},{"alias_kind":"pith_short_16","alias_value":"SE7CS5GLV7Q2R5LF","created_at":"2026-07-05T00:49:52.990632+00:00"},{"alias_kind":"pith_short_8","alias_value":"SE7CS5GL","created_at":"2026-07-05T00:49:52.990632+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12215","citing_title":"MLT-Dedup: Efficient Large-Scale Online Video Deduplication via Multi-Level Representations and Spatial-Temporal Matching","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07775","citing_title":"DALE-CT: Depth-Aware Foundation Models for Computed Tomography","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00958","citing_title":"LeNEPA: No-Augmentation Next-Latent Prediction for Time-Series Representation Learning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06624","citing_title":"Principles and Practice of Deep Representation Learning: or a Mathematical Theory of Memory","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05068","citing_title":"MaCo-GAN: Manifold-Contrastive Adversarial Learning for Single Image Super-Resolution","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01443","citing_title":"UR-JEPA: Uniform Rectifiability as a Regularizer for Joint-Embedding Predictive Architectures","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01971","citing_title":"ProtoFair: Fair Self-Supervised Contrastive Learning via Pseudo-Counterfactual Pairs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17978","citing_title":"MoCo-AIS: A Contrastive Learning Framework for Similarity Computation of Vessel Trajectories","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26059","citing_title":"A welding penetration prediction model for laser welding process based on self-supervised learning using physics-informed neural networks","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2510.16416","citing_title":"SSL4RL: Revisiting Self-supervised Learning as Intrinsic Reward for Visual-Language Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18390","citing_title":"Vision Foundation Models as Generalist Tokenizers for Image Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2002.05709","citing_title":"A Simple Framework for Contrastive Learning of Visual Representations","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2003.04297","citing_title":"Improved Baselines with Momentum Contrastive Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11870","citing_title":"Information theoretic underpinning of self-supervised learning by clustering","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08471","citing_title":"Revisiting Feature Prediction for Learning Visual Representations from Video","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26057","citing_title":"Similarity Choice and Negative Scaling in Supervised Contrastive Learning for Deepfake Audio Detection","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04527","citing_title":"Velox: Learning Representations of 4D Geometry and Appearance","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01971","citing_title":"ProtoFair: Fair Self-Supervised Contrastive Learning via Pseudo-Counterfactual Pairs","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF","json":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF.json","graph_json":"https://pith.science/api/pith-number/SE7CS5GLV7Q2R5LFUOIDRLC7AF/graph.json","events_json":"https://pith.science/api/pith-number/SE7CS5GLV7Q2R5LFUOIDRLC7AF/events.json","paper":"https://pith.science/paper/SE7CS5GL"},"agent_actions":{"view_html":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF","download_json":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF.json","view_paper":"https://pith.science/paper/SE7CS5GL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.05722&json=true","fetch_graph":"https://pith.science/api/pith-number/SE7CS5GLV7Q2R5LFUOIDRLC7AF/graph.json","fetch_events":"https://pith.science/api/pith-number/SE7CS5GLV7Q2R5LFUOIDRLC7AF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF/action/storage_attestation","attest_author":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF/action/author_attestation","sign_citation":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF/action/citation_signature","submit_replication":"https://pith.science/pith/SE7CS5GLV7Q2R5LFUOIDRLC7AF/action/replication_record"}},"created_at":"2026-07-05T00:49:52.990632+00:00","updated_at":"2026-07-05T00:49:52.990632+00:00"}