{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WSLXEFE5THR6WPVZC7M3A55QEV","short_pith_number":"pith:WSLXEFE5","schema_version":"1.0","canonical_sha256":"b49772149d99e3eb3eb917d9b077b025652ac9d97611abe3071d782fcb0a3f81","source":{"kind":"arxiv","id":"2206.07568","version":2},"attestation_state":"computed","paper":{"title":"Contrastive Learning as Goal-Conditioned Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Benjamin Eysenbach, Ruslan Salakhutdinov, Sergey Levine, Tianjun Zhang","submitted_at":"2022-06-15T14:34:15Z","abstract_excerpt":"In reinforcement learning (RL), it is easier to solve a task if given a good representation. While deep RL should automatically acquire such good representations, prior work often finds that learning representations in an end-to-end fashion is unstable and instead equip RL algorithms with additional representation learning parts (e.g., auxiliary losses, data augmentation). How can we design RL algorithms that directly acquire good representations? In this paper, instead of adding representation learning parts to an existing RL algorithm, we show (contrastive) representation learning methods ca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.07568","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-15T14:34:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cd1467383d98ae7cb86f7a0eefbd7149d04978f38f08d14a2ecd67797f3cef13","abstract_canon_sha256":"44bb76acd2c0d65d8ea3f9d389adbc9d3bd0ed4ab41d72624a2d68179a20bc0f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:43:12.017887Z","signature_b64":"SA6x+2Z9/RsMLcD9KOQdHJvdbeQDX9DxApG6qDZQdp8MmaeGStcPdZdQ5aYQXci59lTli+U7InFhIDG3VVgHAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b49772149d99e3eb3eb917d9b077b025652ac9d97611abe3071d782fcb0a3f81","last_reissued_at":"2026-07-05T05:43:12.017464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:43:12.017464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Contrastive Learning as Goal-Conditioned Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Benjamin Eysenbach, Ruslan Salakhutdinov, Sergey Levine, Tianjun Zhang","submitted_at":"2022-06-15T14:34:15Z","abstract_excerpt":"In reinforcement learning (RL), it is easier to solve a task if given a good representation. While deep RL should automatically acquire such good representations, prior work often finds that learning representations in an end-to-end fashion is unstable and instead equip RL algorithms with additional representation learning parts (e.g., auxiliary losses, data augmentation). How can we design RL algorithms that directly acquire good representations? In this paper, instead of adding representation learning parts to an existing RL algorithm, we show (contrastive) representation learning methods ca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.07568","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.07568/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.07568","created_at":"2026-07-05T05:43:12.017524+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.07568v2","created_at":"2026-07-05T05:43:12.017524+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.07568","created_at":"2026-07-05T05:43:12.017524+00:00"},{"alias_kind":"pith_short_12","alias_value":"WSLXEFE5THR6","created_at":"2026-07-05T05:43:12.017524+00:00"},{"alias_kind":"pith_short_16","alias_value":"WSLXEFE5THR6WPVZ","created_at":"2026-07-05T05:43:12.017524+00:00"},{"alias_kind":"pith_short_8","alias_value":"WSLXEFE5","created_at":"2026-07-05T05:43:12.017524+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22711","citing_title":"Abstraction for Offline Goal-Conditioned Reinforcement Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2602.19532","citing_title":"Bellman Value Decomposition for Task Logic in Safe Optimal Control","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08512","citing_title":"MoMo: Conditioned Contrastive Representation Learning for Preference-Modulated Planning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2210.00030","citing_title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09364","citing_title":"Multi-scale Predictive Representations for Goal-conditioned Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08512","citing_title":"MoMo: Conditioned Contrastive Representation Learning for Preference-Modulated Planning","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV","json":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV.json","graph_json":"https://pith.science/api/pith-number/WSLXEFE5THR6WPVZC7M3A55QEV/graph.json","events_json":"https://pith.science/api/pith-number/WSLXEFE5THR6WPVZC7M3A55QEV/events.json","paper":"https://pith.science/paper/WSLXEFE5"},"agent_actions":{"view_html":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV","download_json":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV.json","view_paper":"https://pith.science/paper/WSLXEFE5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.07568&json=true","fetch_graph":"https://pith.science/api/pith-number/WSLXEFE5THR6WPVZC7M3A55QEV/graph.json","fetch_events":"https://pith.science/api/pith-number/WSLXEFE5THR6WPVZC7M3A55QEV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV/action/storage_attestation","attest_author":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV/action/author_attestation","sign_citation":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV/action/citation_signature","submit_replication":"https://pith.science/pith/WSLXEFE5THR6WPVZC7M3A55QEV/action/replication_record"}},"created_at":"2026-07-05T05:43:12.017524+00:00","updated_at":"2026-07-05T05:43:12.017524+00:00"}