{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GHFNAE54SYIAC3R7XSHK6XDRPG","short_pith_number":"pith:GHFNAE54","schema_version":"1.0","canonical_sha256":"31cad013bc9610016e3fbc8eaf5c71798f8a456fc77f20e7ea69baaabb187e74","source":{"kind":"arxiv","id":"2302.02011","version":1},"attestation_state":"computed","paper":{"title":"Zero-Shot Robot Manipulation from Passive Human Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Abhinav Gupta, Homanga Bharadhwaj, Shubham Tulsiani, Vikash Kumar","submitted_at":"2023-02-03T21:39:52Z","abstract_excerpt":"Can we learn robot manipulation for everyday tasks, only by watching videos of humans doing arbitrary tasks in different unstructured settings? Unlike widely adopted strategies of learning task-specific behaviors or direct imitation of a human video, we develop a a framework for extracting agent-agnostic action representations from human videos, and then map it to the agent's embodiment during deployment. Our framework is based on predicting plausible human hand trajectories given an initial image of a scene. After training this prediction model on a diverse set of human videos from the intern"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.02011","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-02-03T21:39:52Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"afec904a88671dec1157c387f9e4246089e7a63d63bd1e78fcca09b54f50408a","abstract_canon_sha256":"7fea8a88a3efb7c0df3607399dd491eca4d62a5ffcac19aa1ba08a441a64175b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:38:47.741371Z","signature_b64":"6NSzitVZhI6ti2OG4gPVR0KdId6wNdr378JQGD17y5hy3Uaa5few75GTsxuBV0lht/Blu4WE53eiONBFaDw2Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31cad013bc9610016e3fbc8eaf5c71798f8a456fc77f20e7ea69baaabb187e74","last_reissued_at":"2026-07-05T05:38:47.740835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:38:47.740835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero-Shot Robot Manipulation from Passive Human Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Abhinav Gupta, Homanga Bharadhwaj, Shubham Tulsiani, Vikash Kumar","submitted_at":"2023-02-03T21:39:52Z","abstract_excerpt":"Can we learn robot manipulation for everyday tasks, only by watching videos of humans doing arbitrary tasks in different unstructured settings? Unlike widely adopted strategies of learning task-specific behaviors or direct imitation of a human video, we develop a a framework for extracting agent-agnostic action representations from human videos, and then map it to the agent's embodiment during deployment. Our framework is based on predicting plausible human hand trajectories given an initial image of a scene. After training this prediction model on a diverse set of human videos from the intern"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.02011","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.02011/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.02011","created_at":"2026-07-05T05:38:47.740894+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.02011v1","created_at":"2026-07-05T05:38:47.740894+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.02011","created_at":"2026-07-05T05:38:47.740894+00:00"},{"alias_kind":"pith_short_12","alias_value":"GHFNAE54SYIA","created_at":"2026-07-05T05:38:47.740894+00:00"},{"alias_kind":"pith_short_16","alias_value":"GHFNAE54SYIAC3R7","created_at":"2026-07-05T05:38:47.740894+00:00"},{"alias_kind":"pith_short_8","alias_value":"GHFNAE54","created_at":"2026-07-05T05:38:47.740894+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06988","citing_title":"WAM-TTT: Steering World-Action Models by Watching Human Play at Test Time","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21672","citing_title":"Imitation from Heterogeneous Demonstrations using Grounded Latent-Action World Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00990","citing_title":"Robotic Manipulation by Imitating Generated Videos Without Physical Demonstrations","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04671","citing_title":"X-Diffusion: Training Diffusion Policies on Cross-Embodiment Human Demonstrations","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15659","citing_title":"FLARE: Robot Learning with Implicit World Modeling","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2401.00025","citing_title":"Any-point Trajectory Modeling for Policy Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12705","citing_title":"DreamGen: Unlocking Generalization in Robot Learning through Video World Models","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2307.05973","citing_title":"VoxPoser: Composable 3D Value Maps for Robotic Manipulation with Language Models","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05756","citing_title":"MaMi-HOI: Harmonizing Global Kinematics and Local Geometry for Human-Object Interaction Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10809","citing_title":"WARPED: Wrist-Aligned Rendering for Robot Policy Learning from Egocentric Human Demonstrations","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG","json":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG.json","graph_json":"https://pith.science/api/pith-number/GHFNAE54SYIAC3R7XSHK6XDRPG/graph.json","events_json":"https://pith.science/api/pith-number/GHFNAE54SYIAC3R7XSHK6XDRPG/events.json","paper":"https://pith.science/paper/GHFNAE54"},"agent_actions":{"view_html":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG","download_json":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG.json","view_paper":"https://pith.science/paper/GHFNAE54","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.02011&json=true","fetch_graph":"https://pith.science/api/pith-number/GHFNAE54SYIAC3R7XSHK6XDRPG/graph.json","fetch_events":"https://pith.science/api/pith-number/GHFNAE54SYIAC3R7XSHK6XDRPG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG/action/storage_attestation","attest_author":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG/action/author_attestation","sign_citation":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG/action/citation_signature","submit_replication":"https://pith.science/pith/GHFNAE54SYIAC3R7XSHK6XDRPG/action/replication_record"}},"created_at":"2026-07-05T05:38:47.740894+00:00","updated_at":"2026-07-05T05:38:47.740894+00:00"}