{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:RNSC2K4ATR3V2G2YOED2R7GLK4","short_pith_number":"pith:RNSC2K4A","schema_version":"1.0","canonical_sha256":"8b642d2b809c775d1b587107a8fccb5704b13f5ef383ce2c88c6b628632f5edd","source":{"kind":"arxiv","id":"2207.09450","version":1},"attestation_state":"computed","paper":{"title":"Human-to-Robot Imitation in the Wild","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Abhinav Gupta, Deepak Pathak, Shikhar Bahl","submitted_at":"2022-07-19T17:59:59Z","abstract_excerpt":"We approach the problem of learning by watching humans in the wild. While traditional approaches in Imitation and Reinforcement Learning are promising for learning in the real world, they are either sample inefficient or are constrained to lab settings. Meanwhile, there has been a lot of success in processing passive, unstructured human data. We propose tackling this problem via an efficient one-shot robot learning algorithm, centered around learning from a third-person perspective. We call our method WHIRL: In-the-Wild Human Imitating Robot Learning. WHIRL extracts a prior over the intent of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.09450","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2022-07-19T17:59:59Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"b74d4b1ab45ed0cccdc7bb823b9b3eb49142836939cff7eb64f03b2a13ee4cdc","abstract_canon_sha256":"5e9d9f3fb64716096b361459c5dace652e87f30adf3af0849bad292ff5a764be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:41:49.752384Z","signature_b64":"rKunKW6bC17YHbZg7MvICO4wR2DteOOUJMNnBEZynJfaGg12bwQAiWxSJaIPxjAd+lcZLQI1l2qF8mBu/WCYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b642d2b809c775d1b587107a8fccb5704b13f5ef383ce2c88c6b628632f5edd","last_reissued_at":"2026-07-05T04:41:49.752020Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:41:49.752020Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Human-to-Robot Imitation in the Wild","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Abhinav Gupta, Deepak Pathak, Shikhar Bahl","submitted_at":"2022-07-19T17:59:59Z","abstract_excerpt":"We approach the problem of learning by watching humans in the wild. While traditional approaches in Imitation and Reinforcement Learning are promising for learning in the real world, they are either sample inefficient or are constrained to lab settings. Meanwhile, there has been a lot of success in processing passive, unstructured human data. We propose tackling this problem via an efficient one-shot robot learning algorithm, centered around learning from a third-person perspective. We call our method WHIRL: In-the-Wild Human Imitating Robot Learning. WHIRL extracts a prior over the intent of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.09450","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.09450/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.09450","created_at":"2026-07-05T04:41:49.752095+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.09450v1","created_at":"2026-07-05T04:41:49.752095+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.09450","created_at":"2026-07-05T04:41:49.752095+00:00"},{"alias_kind":"pith_short_12","alias_value":"RNSC2K4ATR3V","created_at":"2026-07-05T04:41:49.752095+00:00"},{"alias_kind":"pith_short_16","alias_value":"RNSC2K4ATR3V2G2Y","created_at":"2026-07-05T04:41:49.752095+00:00"},{"alias_kind":"pith_short_8","alias_value":"RNSC2K4A","created_at":"2026-07-05T04:41:49.752095+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23685","citing_title":"LaST-HD: Learning Latent Physical Reasoning from Scalable Human Data for Robot Manipulation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21406","citing_title":"Robot Self-Improvement via Human-Video Dynamics Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28133","citing_title":"Translation as a Bridging Action: Transferring Manipulation Skills from Humans to Robots","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":176,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29298","citing_title":"MonoDuo: Using One Robot Arm to Learn Bimanual Policies","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20894","citing_title":"Mobile UMI: Cross-View Diffusion Policy with Decoupled Kinematics for Mobile Manipulation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00990","citing_title":"Robotic Manipulation by Imitating Generated Videos Without Physical Demonstrations","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2210.00030","citing_title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02117","citing_title":"Mobile ALOHA: Learning Bimanual Mobile Manipulation with Low-Cost Whole-Body Teleoperation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2307.05973","citing_title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08571","citing_title":"BEACON: Cross-Domain Co-Training of Generative Robot Policies via Best-Effort Adaptation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08571","citing_title":"BEACON: Cross-Domain Co-Training of Generative Robot Policies via Best-Effort Adaptation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00078","citing_title":"Being-H0.7: A Latent World-Action Model from Egocentric Videos","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07607","citing_title":"EgoVerse: An Egocentric Human Dataset for Robot Learning from Around the World","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2405.12213","citing_title":"Octo: An Open-Source Generalist Robot Policy","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4","json":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4.json","graph_json":"https://pith.science/api/pith-number/RNSC2K4ATR3V2G2YOED2R7GLK4/graph.json","events_json":"https://pith.science/api/pith-number/RNSC2K4ATR3V2G2YOED2R7GLK4/events.json","paper":"https://pith.science/paper/RNSC2K4A"},"agent_actions":{"view_html":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4","download_json":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4.json","view_paper":"https://pith.science/paper/RNSC2K4A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.09450&json=true","fetch_graph":"https://pith.science/api/pith-number/RNSC2K4ATR3V2G2YOED2R7GLK4/graph.json","fetch_events":"https://pith.science/api/pith-number/RNSC2K4ATR3V2G2YOED2R7GLK4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4/action/storage_attestation","attest_author":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4/action/author_attestation","sign_citation":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4/action/citation_signature","submit_replication":"https://pith.science/pith/RNSC2K4ATR3V2G2YOED2R7GLK4/action/replication_record"}},"created_at":"2026-07-05T04:41:49.752095+00:00","updated_at":"2026-07-05T04:41:49.752095+00:00"}