{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:7NPUSCXTOT4UG56XQV7BSUYN37","short_pith_number":"pith:7NPUSCXT","schema_version":"1.0","canonical_sha256":"fb5f490af374f94377d7857e19530ddfff8b479f8a99f513ccbac5ee47d71727","source":{"kind":"arxiv","id":"2206.01342","version":3},"attestation_state":"computed","paper":{"title":"Understanding the Role of Nonlinearity in Training Dynamics of Contrastive Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Yuandong Tian","submitted_at":"2022-06-02T23:52:35Z","abstract_excerpt":"While the empirical success of self-supervised learning (SSL) heavily relies on the usage of deep nonlinear models, existing theoretical works on SSL understanding still focus on linear ones. In this paper, we study the role of nonlinearity in the training dynamics of contrastive learning (CL) on one and two-layer nonlinear networks with homogeneous activation $h(x) = h'(x)x$. We have two major theoretical discoveries. First, the presence of nonlinearity can lead to many local optima even in 1-layer setting, each corresponding to certain patterns from the data distribution, while with linear a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.01342","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-02T23:52:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bee6690cae6f868ea07852fa85d3989dcb644f4bacd88854932a04af654f83b8","abstract_canon_sha256":"2b80e0deed40792fc8bcc374a4eca507b911d057c75d51be3eb5f184c856270d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:47:33.139397Z","signature_b64":"+Cm6JNuXz0mz41ZXL9kIrQMxf145LMFDPIVySyrJGxElhohH/Fd7G4tjBgcX69LmiS4K8s3qXSxuy+JQhB9xDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb5f490af374f94377d7857e19530ddfff8b479f8a99f513ccbac5ee47d71727","last_reissued_at":"2026-07-05T05:47:33.138898Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:47:33.138898Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding the Role of Nonlinearity in Training Dynamics of Contrastive Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Yuandong Tian","submitted_at":"2022-06-02T23:52:35Z","abstract_excerpt":"While the empirical success of self-supervised learning (SSL) heavily relies on the usage of deep nonlinear models, existing theoretical works on SSL understanding still focus on linear ones. In this paper, we study the role of nonlinearity in the training dynamics of contrastive learning (CL) on one and two-layer nonlinear networks with homogeneous activation $h(x) = h'(x)x$. We have two major theoretical discoveries. First, the presence of nonlinearity can lead to many local optima even in 1-layer setting, each corresponding to certain patterns from the data distribution, while with linear a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.01342","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.01342/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.01342","created_at":"2026-07-05T05:47:33.138966+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.01342v3","created_at":"2026-07-05T05:47:33.138966+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.01342","created_at":"2026-07-05T05:47:33.138966+00:00"},{"alias_kind":"pith_short_12","alias_value":"7NPUSCXTOT4U","created_at":"2026-07-05T05:47:33.138966+00:00"},{"alias_kind":"pith_short_16","alias_value":"7NPUSCXTOT4UG56X","created_at":"2026-07-05T05:47:33.138966+00:00"},{"alias_kind":"pith_short_8","alias_value":"7NPUSCXT","created_at":"2026-07-05T05:47:33.138966+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.01278","citing_title":"Risk forecasting using Long Short-Term Memory Mixture Density Networks","ref_index":75,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37","json":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37.json","graph_json":"https://pith.science/api/pith-number/7NPUSCXTOT4UG56XQV7BSUYN37/graph.json","events_json":"https://pith.science/api/pith-number/7NPUSCXTOT4UG56XQV7BSUYN37/events.json","paper":"https://pith.science/paper/7NPUSCXT"},"agent_actions":{"view_html":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37","download_json":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37.json","view_paper":"https://pith.science/paper/7NPUSCXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.01342&json=true","fetch_graph":"https://pith.science/api/pith-number/7NPUSCXTOT4UG56XQV7BSUYN37/graph.json","fetch_events":"https://pith.science/api/pith-number/7NPUSCXTOT4UG56XQV7BSUYN37/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37/action/storage_attestation","attest_author":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37/action/author_attestation","sign_citation":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37/action/citation_signature","submit_replication":"https://pith.science/pith/7NPUSCXTOT4UG56XQV7BSUYN37/action/replication_record"}},"created_at":"2026-07-05T05:47:33.138966+00:00","updated_at":"2026-07-05T05:47:33.138966+00:00"}