{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:LEMK5LPSOQIC2HYOIF2MNZFQ2J","short_pith_number":"pith:LEMK5LPS","schema_version":"1.0","canonical_sha256":"5918aeadf274102d1f0e4174c6e4b0d24b0e459284347dd0444d815c54cde2c4","source":{"kind":"arxiv","id":"2202.00161","version":3},"attestation_state":"computed","paper":{"title":"CIC: Contrastive Intrinsic Control for Unsupervised Skill Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aravind Rajeswaran, Denis Yarats, Hao Liu, Michael Laskin, Pieter Abbeel, Xue Bin Peng","submitted_at":"2022-02-01T00:36:29Z","abstract_excerpt":"We introduce Contrastive Intrinsic Control (CIC), an algorithm for unsupervised skill discovery that maximizes the mutual information between state-transitions and latent skill vectors. CIC utilizes contrastive learning between state-transitions and skills to learn behavior embeddings and maximizes the entropy of these embeddings as an intrinsic reward to encourage behavioral diversity. We evaluate our algorithm on the Unsupervised Reinforcement Learning Benchmark, which consists of a long reward-free pre-training phase followed by a short adaptation phase to downstream tasks with extrinsic re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.00161","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-01T00:36:29Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ab9597e3d31e910dc563996a1a96de018bd10ca028f07b533197c9d977cb731d","abstract_canon_sha256":"8199a2c35d6d607ab53f29b4b10b7a4b658e8ef8cdf59bea8a35bedf417f62a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:01.007772Z","signature_b64":"nfP6oUW4WiBUFgfWjaV9/zFl7Kfr1tQ/zqYlzq/QCyKB5nMhLhzOeWJQwVGkFJoN3lfeFBBLjLXdd5NlH+9iCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5918aeadf274102d1f0e4174c6e4b0d24b0e459284347dd0444d815c54cde2c4","last_reissued_at":"2026-07-05T04:10:01.007362Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:01.007362Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CIC: Contrastive Intrinsic Control for Unsupervised Skill Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aravind Rajeswaran, Denis Yarats, Hao Liu, Michael Laskin, Pieter Abbeel, Xue Bin Peng","submitted_at":"2022-02-01T00:36:29Z","abstract_excerpt":"We introduce Contrastive Intrinsic Control (CIC), an algorithm for unsupervised skill discovery that maximizes the mutual information between state-transitions and latent skill vectors. CIC utilizes contrastive learning between state-transitions and skills to learn behavior embeddings and maximizes the entropy of these embeddings as an intrinsic reward to encourage behavioral diversity. We evaluate our algorithm on the Unsupervised Reinforcement Learning Benchmark, which consists of a long reward-free pre-training phase followed by a short adaptation phase to downstream tasks with extrinsic re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.00161","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.00161/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.00161","created_at":"2026-07-05T04:10:01.007417+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.00161v3","created_at":"2026-07-05T04:10:01.007417+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.00161","created_at":"2026-07-05T04:10:01.007417+00:00"},{"alias_kind":"pith_short_12","alias_value":"LEMK5LPSOQIC","created_at":"2026-07-05T04:10:01.007417+00:00"},{"alias_kind":"pith_short_16","alias_value":"LEMK5LPSOQIC2HYO","created_at":"2026-07-05T04:10:01.007417+00:00"},{"alias_kind":"pith_short_8","alias_value":"LEMK5LPS","created_at":"2026-07-05T04:10:01.007417+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09476","citing_title":"Goal Sets, Not Goal States: Queryable Robot Goals through Goal-Set Hindsight Relabeling","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23652","citing_title":"One Policy, Infinite NPCs: Persona-Traceable Shared RL Policies for Scalable Game Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12338","citing_title":"Manifold Sampling via Entropy Maximization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06145","citing_title":"Unifying Goal-Conditioned RL and Unsupervised Skill Learning via Control-Maximization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":103,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J","json":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J.json","graph_json":"https://pith.science/api/pith-number/LEMK5LPSOQIC2HYOIF2MNZFQ2J/graph.json","events_json":"https://pith.science/api/pith-number/LEMK5LPSOQIC2HYOIF2MNZFQ2J/events.json","paper":"https://pith.science/paper/LEMK5LPS"},"agent_actions":{"view_html":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J","download_json":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J.json","view_paper":"https://pith.science/paper/LEMK5LPS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.00161&json=true","fetch_graph":"https://pith.science/api/pith-number/LEMK5LPSOQIC2HYOIF2MNZFQ2J/graph.json","fetch_events":"https://pith.science/api/pith-number/LEMK5LPSOQIC2HYOIF2MNZFQ2J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J/action/storage_attestation","attest_author":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J/action/author_attestation","sign_citation":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J/action/citation_signature","submit_replication":"https://pith.science/pith/LEMK5LPSOQIC2HYOIF2MNZFQ2J/action/replication_record"}},"created_at":"2026-07-05T04:10:01.007417+00:00","updated_at":"2026-07-05T04:10:01.007417+00:00"}