{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:YUQ326UJW3JBST3BPMWFW4FYZJ","short_pith_number":"pith:YUQ326UJ","schema_version":"1.0","canonical_sha256":"c521bd7a89b6d2194f617b2c5b70b8ca564eb4121551344cdd531bd99af7cfe6","source":{"kind":"arxiv","id":"2106.09678","version":1},"attestation_state":"computed","paper":{"title":"SECANT: Self-Expert Cloning for Zero-Shot Generalization of Visual Policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Anima Anandkumar, De-An Huang, Guanzhi Wang, Li Fei-Fei, Linxi Fan, Yuke Zhu, Zhiding Yu","submitted_at":"2021-06-17T17:28:18Z","abstract_excerpt":"Generalization has been a long-standing challenge for reinforcement learning (RL). Visual RL, in particular, can be easily distracted by irrelevant factors in high-dimensional observation space. In this work, we consider robust policy learning which targets zero-shot generalization to unseen visual environments with large distributional shift. We propose SECANT, a novel self-expert cloning technique that leverages image augmentation in two stages to decouple robust representation learning from policy optimization. Specifically, an expert policy is first trained by RL from scratch with weak aug"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.09678","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-17T17:28:18Z","cross_cats_sorted":["cs.AI","cs.CV","cs.RO"],"title_canon_sha256":"8a80e832c087baf3a53532ce4c482ccd3fe86f91bb5f43dbcd1d406d9e05a791","abstract_canon_sha256":"fcd23985d0a745daa85295f4bebf5ca16a5cc2173186b2103bc915acb7b9b9a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:18.615671Z","signature_b64":"wLsDxXEcBoZTi1SSM1nL5HieqanoMgXRn4z5fj7SSQAlWmRW8iPAgj/XZwgiUL3tgo3QPTcvR4NLqPvjIe6XBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c521bd7a89b6d2194f617b2c5b70b8ca564eb4121551344cdd531bd99af7cfe6","last_reissued_at":"2026-07-05T02:50:18.615155Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:18.615155Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SECANT: Self-Expert Cloning for Zero-Shot Generalization of Visual Policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Anima Anandkumar, De-An Huang, Guanzhi Wang, Li Fei-Fei, Linxi Fan, Yuke Zhu, Zhiding Yu","submitted_at":"2021-06-17T17:28:18Z","abstract_excerpt":"Generalization has been a long-standing challenge for reinforcement learning (RL). Visual RL, in particular, can be easily distracted by irrelevant factors in high-dimensional observation space. In this work, we consider robust policy learning which targets zero-shot generalization to unseen visual environments with large distributional shift. We propose SECANT, a novel self-expert cloning technique that leverages image augmentation in two stages to decouple robust representation learning from policy optimization. Specifically, an expert policy is first trained by RL from scratch with weak aug"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.09678","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.09678/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.09678","created_at":"2026-07-05T02:50:18.615217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.09678v1","created_at":"2026-07-05T02:50:18.615217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.09678","created_at":"2026-07-05T02:50:18.615217+00:00"},{"alias_kind":"pith_short_12","alias_value":"YUQ326UJW3JB","created_at":"2026-07-05T02:50:18.615217+00:00"},{"alias_kind":"pith_short_16","alias_value":"YUQ326UJW3JBST3B","created_at":"2026-07-05T02:50:18.615217+00:00"},{"alias_kind":"pith_short_8","alias_value":"YUQ326UJ","created_at":"2026-07-05T02:50:18.615217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2403.09227","citing_title":"BEHAVIOR-1K: A Human-Centered, Embodied AI Benchmark with 1,000 Everyday Activities and Realistic Simulation","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ","json":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ.json","graph_json":"https://pith.science/api/pith-number/YUQ326UJW3JBST3BPMWFW4FYZJ/graph.json","events_json":"https://pith.science/api/pith-number/YUQ326UJW3JBST3BPMWFW4FYZJ/events.json","paper":"https://pith.science/paper/YUQ326UJ"},"agent_actions":{"view_html":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ","download_json":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ.json","view_paper":"https://pith.science/paper/YUQ326UJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.09678&json=true","fetch_graph":"https://pith.science/api/pith-number/YUQ326UJW3JBST3BPMWFW4FYZJ/graph.json","fetch_events":"https://pith.science/api/pith-number/YUQ326UJW3JBST3BPMWFW4FYZJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ/action/storage_attestation","attest_author":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ/action/author_attestation","sign_citation":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ/action/citation_signature","submit_replication":"https://pith.science/pith/YUQ326UJW3JBST3BPMWFW4FYZJ/action/replication_record"}},"created_at":"2026-07-05T02:50:18.615217+00:00","updated_at":"2026-07-05T02:50:18.615217+00:00"}