{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:NJNQRTVZBOPHLDHVHCQGUNVRN4","short_pith_number":"pith:NJNQRTVZ","schema_version":"1.0","canonical_sha256":"6a5b08ceb90b9e758cf538a06a36b16f2ba4fc72aea060987fa098cb0e5fc37e","source":{"kind":"arxiv","id":"1903.03698","version":4},"attestation_state":"computed","paper":{"title":"Skew-Fit: State-Covering Self-Supervised Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashvin Nair, Murtaza Dalal, Sergey Levine, Shikhar Bahl, Steven Lin, Vitchyr H. Pong","submitted_at":"2019-03-08T23:32:17Z","abstract_excerpt":"Autonomous agents that must exhibit flexible and broad capabilities will need to be equipped with large repertoires of skills. Defining each skill with a manually-designed reward function limits this repertoire and imposes a manual engineering burden. Self-supervised agents that set their own goals can automate this process, but designing appropriate goal setting objectives can be difficult, and often involves heuristic design decisions. In this paper, we propose a formal exploration objective for goal-reaching policies that maximizes state coverage. We show that this objective is equivalent t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1903.03698","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-03-08T23:32:17Z","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"title_canon_sha256":"835ab214c8fc4aebce8eb064b28809309f215682a583810f86fa389253a38a93","abstract_canon_sha256":"61303a4a34252d0d79331dc7c1bd486ea239670552d3ec7b72d2a76dbf8d1a88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:24:14.639399Z","signature_b64":"JOHwSEMFdYTnml+dKVoGnxezppUGn9kaguuIYNwV97kRYJpqjuKejgfAe2We0tDJiTaHJFxAWb7B5WXiXD+7Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a5b08ceb90b9e758cf538a06a36b16f2ba4fc72aea060987fa098cb0e5fc37e","last_reissued_at":"2026-07-05T01:24:14.638849Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:24:14.638849Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Skew-Fit: State-Covering Self-Supervised Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashvin Nair, Murtaza Dalal, Sergey Levine, Shikhar Bahl, Steven Lin, Vitchyr H. Pong","submitted_at":"2019-03-08T23:32:17Z","abstract_excerpt":"Autonomous agents that must exhibit flexible and broad capabilities will need to be equipped with large repertoires of skills. Defining each skill with a manually-designed reward function limits this repertoire and imposes a manual engineering burden. Self-supervised agents that set their own goals can automate this process, but designing appropriate goal setting objectives can be difficult, and often involves heuristic design decisions. In this paper, we propose a formal exploration objective for goal-reaching policies that maximizes state coverage. We show that this objective is equivalent t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1903.03698","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1903.03698/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1903.03698","created_at":"2026-07-05T01:24:14.638907+00:00"},{"alias_kind":"arxiv_version","alias_value":"1903.03698v4","created_at":"2026-07-05T01:24:14.638907+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1903.03698","created_at":"2026-07-05T01:24:14.638907+00:00"},{"alias_kind":"pith_short_12","alias_value":"NJNQRTVZBOPH","created_at":"2026-07-05T01:24:14.638907+00:00"},{"alias_kind":"pith_short_16","alias_value":"NJNQRTVZBOPHLDHV","created_at":"2026-07-05T01:24:14.638907+00:00"},{"alias_kind":"pith_short_8","alias_value":"NJNQRTVZ","created_at":"2026-07-05T01:24:14.638907+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21788","citing_title":"Rotation-Aware Point-Cloud Embeddings for Vision-Based In-Hand Reorientation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09476","citing_title":"Goal Sets, Not Goal States: Queryable Robot Goals through Goal-Set Hindsight Relabeling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23551","citing_title":"Goal-Conditioned Agents that Learn Everything All at Once","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2502.12272","citing_title":"Learning to Reason at the Frontier of Learnability","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21039","citing_title":"Strict Subgoal Execution: Reliable Long-Horizon Planning in Hierarchical Reinforcement Learning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14350","citing_title":"Distributionally Robust Multi-Task Reinforcement Learning via Adaptive Task Sampling","ref_index":275,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25496","citing_title":"Improving Zero-Shot Offline RL via Behavioral Task Sampling","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06145","citing_title":"Unifying Goal-Conditioned RL and Unsupervised Skill Learning via Control-Maximization","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01242","citing_title":"Breaking the Computational Barrier: Provably Efficient Actor-Critic for Low-Rank MDPs","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":147,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4","json":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4.json","graph_json":"https://pith.science/api/pith-number/NJNQRTVZBOPHLDHVHCQGUNVRN4/graph.json","events_json":"https://pith.science/api/pith-number/NJNQRTVZBOPHLDHVHCQGUNVRN4/events.json","paper":"https://pith.science/paper/NJNQRTVZ"},"agent_actions":{"view_html":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4","download_json":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4.json","view_paper":"https://pith.science/paper/NJNQRTVZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1903.03698&json=true","fetch_graph":"https://pith.science/api/pith-number/NJNQRTVZBOPHLDHVHCQGUNVRN4/graph.json","fetch_events":"https://pith.science/api/pith-number/NJNQRTVZBOPHLDHVHCQGUNVRN4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4/action/storage_attestation","attest_author":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4/action/author_attestation","sign_citation":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4/action/citation_signature","submit_replication":"https://pith.science/pith/NJNQRTVZBOPHLDHVHCQGUNVRN4/action/replication_record"}},"created_at":"2026-07-05T01:24:14.638907+00:00","updated_at":"2026-07-05T01:24:14.638907+00:00"}