{"work":{"id":"7b08a1d4-d565-424e-9c86-6ef244b7b90a","openalex_id":"https://openalex.org/W4283793451","doi":"10.1609/aaai.v36i10.21390","arxiv_id":"1807.03748","raw_key":null,"title":"Representation Learning with Contrastive Predictive Coding","authors":null,"authors_text":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals","year":2018,"venue":"cs.LG","abstract":"While supervised learning has enabled great progress in many applications, unsupervised learning has not seen such widespread adoption, and remains an important and challenging endeavor for artificial intelligence. In this work, we propose a universal unsupervised learning approach to extract useful representations from high-dimensional data, which we call Contrastive Predictive Coding. The key insight of our model is to learn such representations by predicting the future in latent space by using powerful autoregressive models. We use a probabilistic contrastive loss which induces the latent space to capture information that is maximally useful to predict future samples. It also makes the model tractable by using negative sampling. While most prior work has focused on evaluating representations for a particular modality, we demonstrate that our approach is able to learn useful representations achieving strong performance on four distinct domains: speech, images, text and reinforcement learning in 3D environments.","external_url":"https://arxiv.org/abs/1807.03748","cited_by_count":14,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"1807.03748","created_at":"2026-05-09T00:04:25.666578+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Representation Learning with Contrastive Predictive Coding","render_title":"Representation Learning with Contrastive Predictive Coding"},"hub":{"state":{"work_id":"7b08a1d4-d565-424e-9c86-6ef244b7b90a","tier":"super_hub","tier_reason":"100+ Pith inbound or 10,000+ external citations","pith_inbound_count":597,"external_cited_by_count":14,"distinct_field_count":37,"first_pith_cited_at":"2019-06-21T16:54:42+00:00","last_pith_cited_at":"2026-07-09T13:46:54+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T05:09:21.413705+00:00","tier_text":"super_hub"},"tier":"super_hub","role_counts":[{"context_role":"background","n":30},{"context_role":"method","n":23},{"context_role":"dataset","n":2},{"context_role":"baseline","n":1}],"polarity_counts":[{"context_polarity":"background","n":29},{"context_polarity":"use_method","n":22},{"context_polarity":"unclear","n":2},{"context_polarity":"use_dataset","n":2},{"context_polarity":"baseline","n":1}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"Representation Learning with Contrastive Predictive Coding","claims":[{"claim_text":"While supervised learning has enabled great progress in many applications, unsupervised learning has not seen such widespread adoption, and remains an important and challenging endeavor for artificial intelligence. In this work, we propose a universal unsupervised learning approach to extract useful representations from high-dimensional data, which we call Contrastive Predictive Coding. The key insight of our model is to learn such representations by predicting the future in latent space by using powerful autoregressive models. We use a probabilistic contrastive loss which induces the latent s","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Representation Learning with Contrastive Predictive Coding because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T20:33:34.861480+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"f840049f-2a04-4b06-871b-6a2fd03af217","orcid":null,"display_name":"Aaron van den Oord"},{"id":"42d8048b-5b60-42a3-b126-74f0b795cb25","orcid":null,"display_name":"Yazhe Li"},{"id":"6f9a292e-9a4c-4190-a226-7b1655319ae2","orcid":null,"display_name":"and Oriol Vinyals"}]},"error":null,"updated_at":"2026-05-13T20:33:35.279679+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-13T20:33:30.815668+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":15},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":14},{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":12},{"title":"DINOv3","work_id":"c8b07deb-8fe7-4e18-9620-f3569d3529ce","shared_citers":12},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":11},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":11},{"title":"Qwen3 Embedding: Advancing Text Embedding and Reranking Through Foundation Models","work_id":"bab684a8-d933-426c-a19e-2c855a0d1f59","shared_citers":10},{"title":"Improved Baselines with Momentum Contrastive Learning","work_id":"f275e715-bcdc-487c-bc31-6f98d8a01f5c","shared_citers":8},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":8},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":8},{"title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","shared_citers":8},{"title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","work_id":"50eec732-2d41-432f-9dcf-ac7fff235ea5","shared_citers":8},{"title":"The information bottleneck method","work_id":"72655a80-0724-45ad-a330-1f4ed7aa613b","shared_citers":8},{"title":"A Simple Framework for Contrastive Learning of Visual Representations","work_id":"77d995ce-c44e-4692-9c54-cf8ce771464a","shared_citers":7},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":7},{"title":"Towards General Text Embeddings with Multi-stage Contrastive Learning","work_id":"861a61de-66fe-49d1-b1ab-11f8b082a4cc","shared_citers":7},{"title":"arXiv preprint arXiv:1808.06670 , year=","work_id":"3ec85dc8-3ecb-4425-ae6b-53ff1d8229fd","shared_citers":6},{"title":"Auto-Encoding Variational Bayes","work_id":"97d95295-30e1-42b4-bbf6-85f0fa4edb44","shared_citers":6},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":6},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":6},{"title":"Efficient Estimation of Word Representations in Vector Space","work_id":"59edaa01-a696-45b3-9a08-5eae777a799e","shared_citers":6},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":6},{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","shared_citers":6},{"title":"LoRA: Low-Rank Adaptation of Large Language Models","work_id":"0426219a-789e-4964-adc8-a04538510818","shared_citers":6}],"time_series":[{"n":1,"year":2019},{"n":2,"year":2020},{"n":3,"year":2021},{"n":3,"year":2023},{"n":1,"year":2024},{"n":1,"year":2025},{"n":174,"year":2026}]},"error":null,"updated_at":"2026-05-13T20:33:33.222882+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"fixed":1,"items":[{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-13T20:33:37.526579+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"Representation Learning with Contrastive Predictive Coding","claims":[{"claim_text":"While supervised learning has enabled great progress in many applications, unsupervised learning has not seen such widespread adoption, and remains an important and challenging endeavor for artificial intelligence. In this work, we propose a universal unsupervised learning approach to extract useful representations from high-dimensional data, which we call Contrastive Predictive Coding. The key insight of our model is to learn such representations by predicting the future in latent space by using powerful autoregressive models. We use a probabilistic contrastive loss which induces the latent s","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Representation Learning with Contrastive Predictive Coding because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T20:33:34.858550+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Representation Learning with Contrastive Predictive Coding","claims":[{"claim_text":"While supervised learning has enabled great progress in many applications, unsupervised learning has not seen such widespread adoption, and remains an important and challenging endeavor for artificial intelligence. In this work, we propose a universal unsupervised learning approach to extract useful representations from high-dimensional data, which we call Contrastive Predictive Coding. The key insight of our model is to learn such representations by predicting the future in latent space by using powerful autoregressive models. We use a probabilistic contrastive loss which induces the latent s","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Representation Learning with Contrastive Predictive Coding because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T20:33:33.226798+00:00"}},"summary":{"title":"Representation Learning with Contrastive Predictive Coding","claims":[{"claim_text":"While supervised learning has enabled great progress in many applications, unsupervised learning has not seen such widespread adoption, and remains an important and challenging endeavor for artificial intelligence. In this work, we propose a universal unsupervised learning approach to extract useful representations from high-dimensional data, which we call Contrastive Predictive Coding. The key insight of our model is to learn such representations by predicting the future in latent space by using powerful autoregressive models. We use a probabilistic contrastive loss which induces the latent s","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Representation Learning with Contrastive Predictive Coding because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":15},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":14},{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":12},{"title":"DINOv3","work_id":"c8b07deb-8fe7-4e18-9620-f3569d3529ce","shared_citers":12},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":11},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":11},{"title":"Qwen3 Embedding: Advancing Text Embedding and Reranking Through Foundation Models","work_id":"bab684a8-d933-426c-a19e-2c855a0d1f59","shared_citers":10},{"title":"Improved Baselines with Momentum Contrastive Learning","work_id":"f275e715-bcdc-487c-bc31-6f98d8a01f5c","shared_citers":8},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":8},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":8},{"title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","shared_citers":8},{"title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","work_id":"50eec732-2d41-432f-9dcf-ac7fff235ea5","shared_citers":8},{"title":"The information bottleneck method","work_id":"72655a80-0724-45ad-a330-1f4ed7aa613b","shared_citers":8},{"title":"A Simple Framework for Contrastive Learning of Visual Representations","work_id":"77d995ce-c44e-4692-9c54-cf8ce771464a","shared_citers":7},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":7},{"title":"Towards General Text Embeddings with Multi-stage Contrastive Learning","work_id":"861a61de-66fe-49d1-b1ab-11f8b082a4cc","shared_citers":7},{"title":"arXiv preprint arXiv:1808.06670 , year=","work_id":"3ec85dc8-3ecb-4425-ae6b-53ff1d8229fd","shared_citers":6},{"title":"Auto-Encoding Variational Bayes","work_id":"97d95295-30e1-42b4-bbf6-85f0fa4edb44","shared_citers":6},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":6},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":6},{"title":"Efficient Estimation of Word Representations in Vector Space","work_id":"59edaa01-a696-45b3-9a08-5eae777a799e","shared_citers":6},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":6},{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","shared_citers":6},{"title":"LoRA: Low-Rank Adaptation of Large Language Models","work_id":"0426219a-789e-4964-adc8-a04538510818","shared_citers":6}],"time_series":[{"n":1,"year":2019},{"n":2,"year":2020},{"n":3,"year":2021},{"n":3,"year":2023},{"n":1,"year":2024},{"n":1,"year":2025},{"n":174,"year":2026}]},"authors":[{"id":"f840049f-2a04-4b06-871b-6a2fd03af217","orcid":null,"display_name":"Aaron van den Oord","source":"manual","import_confidence":0.72},{"id":"6f9a292e-9a4c-4190-a226-7b1655319ae2","orcid":null,"display_name":"and Oriol Vinyals","source":"manual","import_confidence":0.72},{"id":"42d8048b-5b60-42a3-b126-74f0b795cb25","orcid":null,"display_name":"Yazhe Li","source":"manual","import_confidence":0.72}]}}