{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:YOJW7Y3OE65D7SXXPCIZJIFKDI","short_pith_number":"pith:YOJW7Y3O","schema_version":"1.0","canonical_sha256":"c3936fe36e27ba3fcaf7789194a0aa1a30e46e6d3b1e0f963543c16b8297b70d","source":{"kind":"arxiv","id":"2209.14935","version":2},"attestation_state":"computed","paper":{"title":"Does Zero-Shot Reinforcement Learning Exist?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ahmed Touati, J\\'er\\'emy Rapin, Yann Ollivier","submitted_at":"2022-09-29T16:54:05Z","abstract_excerpt":"A zero-shot RL agent is an agent that can solve any RL task in a given environment, instantly with no additional planning or learning, after an initial reward-free learning phase. This marks a shift from the reward-centric RL paradigm towards \"controllable\" agents that can follow arbitrary instructions in an environment. Current RL agents can solve families of related tasks at best, or require planning anew for each task. Strategies for approximate zero-shot RL ave been suggested using successor features (SFs) [BBQ+ 18] or forward-backward (FB) representations [TO21], but testing has been limi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.14935","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-09-29T16:54:05Z","cross_cats_sorted":[],"title_canon_sha256":"d21237c6ee638ae0e10d321108dfb628adf41b98e98fa54fb37667fe5be1c1c7","abstract_canon_sha256":"e8b8938eb754d35a8435e72ef855bc33b2ee0e6d9b54cc2398d037a851651322"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:47:04.118835Z","signature_b64":"nzt6sAdT1aSUIa6AMygJFPwGhdqoYm3paJR3H0DNiEvl7HlYZalDvRvpfuiGAj/k4MA3SZNXzkwQMOFem2AyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3936fe36e27ba3fcaf7789194a0aa1a30e46e6d3b1e0f963543c16b8297b70d","last_reissued_at":"2026-07-05T05:47:04.118327Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:47:04.118327Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does Zero-Shot Reinforcement Learning Exist?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ahmed Touati, J\\'er\\'emy Rapin, Yann Ollivier","submitted_at":"2022-09-29T16:54:05Z","abstract_excerpt":"A zero-shot RL agent is an agent that can solve any RL task in a given environment, instantly with no additional planning or learning, after an initial reward-free learning phase. This marks a shift from the reward-centric RL paradigm towards \"controllable\" agents that can follow arbitrary instructions in an environment. Current RL agents can solve families of related tasks at best, or require planning anew for each task. Strategies for approximate zero-shot RL ave been suggested using successor features (SFs) [BBQ+ 18] or forward-backward (FB) representations [TO21], but testing has been limi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.14935","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.14935/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.14935","created_at":"2026-07-05T05:47:04.118387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.14935v2","created_at":"2026-07-05T05:47:04.118387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.14935","created_at":"2026-07-05T05:47:04.118387+00:00"},{"alias_kind":"pith_short_12","alias_value":"YOJW7Y3OE65D","created_at":"2026-07-05T05:47:04.118387+00:00"},{"alias_kind":"pith_short_16","alias_value":"YOJW7Y3OE65D7SXX","created_at":"2026-07-05T05:47:04.118387+00:00"},{"alias_kind":"pith_short_8","alias_value":"YOJW7Y3O","created_at":"2026-07-05T05:47:04.118387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17017","citing_title":"When Dynamics Shift, Robust Task Inference Wins: Offline Imitation Learning with Behavior Foundation Models Revisited","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20103","citing_title":"Spectral Alignment in Forward-Backward Representations via Temporal Abstraction","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25496","citing_title":"Improving Zero-Shot Offline RL via Behavioral Task Sampling","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI","json":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI.json","graph_json":"https://pith.science/api/pith-number/YOJW7Y3OE65D7SXXPCIZJIFKDI/graph.json","events_json":"https://pith.science/api/pith-number/YOJW7Y3OE65D7SXXPCIZJIFKDI/events.json","paper":"https://pith.science/paper/YOJW7Y3O"},"agent_actions":{"view_html":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI","download_json":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI.json","view_paper":"https://pith.science/paper/YOJW7Y3O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.14935&json=true","fetch_graph":"https://pith.science/api/pith-number/YOJW7Y3OE65D7SXXPCIZJIFKDI/graph.json","fetch_events":"https://pith.science/api/pith-number/YOJW7Y3OE65D7SXXPCIZJIFKDI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI/action/storage_attestation","attest_author":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI/action/author_attestation","sign_citation":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI/action/citation_signature","submit_replication":"https://pith.science/pith/YOJW7Y3OE65D7SXXPCIZJIFKDI/action/replication_record"}},"created_at":"2026-07-05T05:47:04.118387+00:00","updated_at":"2026-07-05T05:47:04.118387+00:00"}