{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:QZLGNTTTSDSKIGHK5YTPR2VFDC","short_pith_number":"pith:QZLGNTTT","schema_version":"1.0","canonical_sha256":"865666ce7390e4a418eaee26f8eaa5189fa78e7c190391b4bcfa29d0c9997ea6","source":{"kind":"arxiv","id":"2004.04136","version":4},"attestation_state":"computed","paper":{"title":"CURL: Contrastive Unsupervised Representations for Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aravind Srinivas, Michael Laskin, Pieter Abbeel","submitted_at":"2020-04-08T17:40:43Z","abstract_excerpt":"We present CURL: Contrastive Unsupervised Representations for Reinforcement Learning. CURL extracts high-level features from raw pixels using contrastive learning and performs off-policy control on top of the extracted features. CURL outperforms prior pixel-based methods, both model-based and model-free, on complex tasks in the DeepMind Control Suite and Atari Games showing 1.9x and 1.2x performance gains at the 100K environment and interaction steps benchmarks respectively. On the DeepMind Control Suite, CURL is the first image-based algorithm to nearly match the sample-efficiency of methods "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.04136","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-08T17:40:43Z","cross_cats_sorted":["cs.CV","stat.ML"],"title_canon_sha256":"da213d06ab0c66e0729d2ce89a6d9bf1ee06bc74d9c8c580cd2ab7303116a4a6","abstract_canon_sha256":"4c79d5852bdc673c99e51d6f9bcf63a40a882471c185b7b3e7b900f8ef92f583"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:36:50.125491Z","signature_b64":"aQT1R2yq30D03EHcnrmugyDYvNd72gMbv/Qu7o0b1xsO86yOl9tW/V2BG21ZaPUZb5+Q3Jprf3K++PjcYrQeAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"865666ce7390e4a418eaee26f8eaa5189fa78e7c190391b4bcfa29d0c9997ea6","last_reissued_at":"2026-07-05T01:36:50.125037Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:36:50.125037Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CURL: Contrastive Unsupervised Representations for Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aravind Srinivas, Michael Laskin, Pieter Abbeel","submitted_at":"2020-04-08T17:40:43Z","abstract_excerpt":"We present CURL: Contrastive Unsupervised Representations for Reinforcement Learning. CURL extracts high-level features from raw pixels using contrastive learning and performs off-policy control on top of the extracted features. CURL outperforms prior pixel-based methods, both model-based and model-free, on complex tasks in the DeepMind Control Suite and Atari Games showing 1.9x and 1.2x performance gains at the 100K environment and interaction steps benchmarks respectively. On the DeepMind Control Suite, CURL is the first image-based algorithm to nearly match the sample-efficiency of methods "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.04136","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.04136/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.04136","created_at":"2026-07-05T01:36:50.125093+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.04136v4","created_at":"2026-07-05T01:36:50.125093+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.04136","created_at":"2026-07-05T01:36:50.125093+00:00"},{"alias_kind":"pith_short_12","alias_value":"QZLGNTTTSDSK","created_at":"2026-07-05T01:36:50.125093+00:00"},{"alias_kind":"pith_short_16","alias_value":"QZLGNTTTSDSKIGHK","created_at":"2026-07-05T01:36:50.125093+00:00"},{"alias_kind":"pith_short_8","alias_value":"QZLGNTTT","created_at":"2026-07-05T01:36:50.125093+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22082","citing_title":"CoRMA: Contrastive RMA for Contact-Rich Meta-Adaptation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26478","citing_title":"Efficient On-policy Visual-RL via Stochastic Decoupled Policy Gradient","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22082","citing_title":"CoRMA: Contrastive RMA for Contact-Rich Meta-Adaptation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2010.02193","citing_title":"Mastering Atari with Discrete World Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2310.16828","citing_title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11291","citing_title":"Optimal Representations for Generalized Contrastive Learning with Imbalanced Datasets","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09364","citing_title":"Multi-scale Predictive Representations for Goal-conditioned Reinforcement Learning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":175,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC","json":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC.json","graph_json":"https://pith.science/api/pith-number/QZLGNTTTSDSKIGHK5YTPR2VFDC/graph.json","events_json":"https://pith.science/api/pith-number/QZLGNTTTSDSKIGHK5YTPR2VFDC/events.json","paper":"https://pith.science/paper/QZLGNTTT"},"agent_actions":{"view_html":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC","download_json":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC.json","view_paper":"https://pith.science/paper/QZLGNTTT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.04136&json=true","fetch_graph":"https://pith.science/api/pith-number/QZLGNTTTSDSKIGHK5YTPR2VFDC/graph.json","fetch_events":"https://pith.science/api/pith-number/QZLGNTTTSDSKIGHK5YTPR2VFDC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC/action/storage_attestation","attest_author":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC/action/author_attestation","sign_citation":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC/action/citation_signature","submit_replication":"https://pith.science/pith/QZLGNTTTSDSKIGHK5YTPR2VFDC/action/replication_record"}},"created_at":"2026-07-05T01:36:50.125093+00:00","updated_at":"2026-07-05T01:36:50.125093+00:00"}