{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5GNANMHDQIWTVXKOEKCDLMMSBL","short_pith_number":"pith:5GNANMHD","schema_version":"1.0","canonical_sha256":"e99a06b0e3822d3add4e228435b1920acb96b3d46e5469b28af5c8887599cd4a","source":{"kind":"arxiv","id":"2407.00143","version":2},"attestation_state":"computed","paper":{"title":"InfoNCE: Identifying the Gap Between Theory and Practice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Attila Juhos, Evgenia Rusak, Oliver Bringmann, Patrik Reizinger, Roland S. Zimmermann, Wieland Brendel","submitted_at":"2024-06-28T16:08:26Z","abstract_excerpt":"Prior theory work on Contrastive Learning via the InfoNCE loss showed that, under certain assumptions, the learned representations recover the ground-truth latent factors. We argue that these theories overlook crucial aspects of how CL is deployed in practice. Specifically, they either assume equal variance across all latents or that certain latents are kept invariant. However, in practice, positive pairs are often generated using augmentations such as strong cropping to just a few pixels. Hence, a more realistic assumption is that all latent factors change with a continuum of variability acro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00143","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-28T16:08:26Z","cross_cats_sorted":["cs.CV","stat.ML"],"title_canon_sha256":"c6f222981592dfda7ca2627eb70d962a1f17e6c6bce13e3c86bbf678cf6b7e59","abstract_canon_sha256":"19e0a958afcd48628ed58c1d2e1e24df090781a368bf96553b23ce231ff1925e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:49:45.212916Z","signature_b64":"W2pKjJH3HChFNs/Nk2/keTtGcfQaapcBspwfgj760Fc9uyYbh7aPy9KcIjC89f3EOfw7oC2nESMsWE+75QCWDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e99a06b0e3822d3add4e228435b1920acb96b3d46e5469b28af5c8887599cd4a","last_reissued_at":"2026-07-05T10:49:45.212434Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:49:45.212434Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InfoNCE: Identifying the Gap Between Theory and Practice","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Attila Juhos, Evgenia Rusak, Oliver Bringmann, Patrik Reizinger, Roland S. Zimmermann, Wieland Brendel","submitted_at":"2024-06-28T16:08:26Z","abstract_excerpt":"Prior theory work on Contrastive Learning via the InfoNCE loss showed that, under certain assumptions, the learned representations recover the ground-truth latent factors. We argue that these theories overlook crucial aspects of how CL is deployed in practice. Specifically, they either assume equal variance across all latents or that certain latents are kept invariant. However, in practice, positive pairs are often generated using augmentations such as strong cropping to just a few pixels. Hence, a more realistic assumption is that all latent factors change with a continuum of variability acro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00143","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00143","created_at":"2026-07-05T10:49:45.212492+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00143v2","created_at":"2026-07-05T10:49:45.212492+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00143","created_at":"2026-07-05T10:49:45.212492+00:00"},{"alias_kind":"pith_short_12","alias_value":"5GNANMHDQIWT","created_at":"2026-07-05T10:49:45.212492+00:00"},{"alias_kind":"pith_short_16","alias_value":"5GNANMHDQIWTVXKO","created_at":"2026-07-05T10:49:45.212492+00:00"},{"alias_kind":"pith_short_8","alias_value":"5GNANMHD","created_at":"2026-07-05T10:49:45.212492+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26379","citing_title":"When Does LeJEPA Learn a World Model?","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14752","citing_title":"X-Tokenizer: A Multimodal Action Tokenizer for Vision-Language-Action Pretraining","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25334","citing_title":"Dual-Pathway Geometry-Aware MLLM for Spatial Intelligence","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2405.10498","citing_title":"A Deep Learning Approach to Heterogeneous Consumer Aesthetics in Fast Fashion","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL","json":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL.json","graph_json":"https://pith.science/api/pith-number/5GNANMHDQIWTVXKOEKCDLMMSBL/graph.json","events_json":"https://pith.science/api/pith-number/5GNANMHDQIWTVXKOEKCDLMMSBL/events.json","paper":"https://pith.science/paper/5GNANMHD"},"agent_actions":{"view_html":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL","download_json":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL.json","view_paper":"https://pith.science/paper/5GNANMHD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00143&json=true","fetch_graph":"https://pith.science/api/pith-number/5GNANMHDQIWTVXKOEKCDLMMSBL/graph.json","fetch_events":"https://pith.science/api/pith-number/5GNANMHDQIWTVXKOEKCDLMMSBL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL/action/storage_attestation","attest_author":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL/action/author_attestation","sign_citation":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL/action/citation_signature","submit_replication":"https://pith.science/pith/5GNANMHDQIWTVXKOEKCDLMMSBL/action/replication_record"}},"created_at":"2026-07-05T10:49:45.212492+00:00","updated_at":"2026-07-05T10:49:45.212492+00:00"}