{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OJZ3WRQZTAODU2V7GECN4446OE","short_pith_number":"pith:OJZ3WRQZ","schema_version":"1.0","canonical_sha256":"7273bb4619981c3a6abf3104de739e713001df26874a605315e572d39c80e2d7","source":{"kind":"arxiv","id":"2404.15709","version":3},"attestation_state":"computed","paper":{"title":"ViViDex: Learning Vision-based Dexterous Manipulation from Human Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Etienne Arlaud, Ivan Laptev, Shizhe Chen, Zerui Chen","submitted_at":"2024-04-24T07:58:28Z","abstract_excerpt":"In this work, we aim to learn a unified vision-based policy for multi-fingered robot hands to manipulate a variety of objects in diverse poses. Though prior work has shown benefits of using human videos for policy learning, performance gains have been limited by the noise in estimated trajectories. Moreover, reliance on privileged object information such as ground-truth object states further limits the applicability in realistic scenarios. To address these limitations, we propose a new framework ViViDex to improve vision-based policy learning from human videos. It first uses reinforcement lear"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.15709","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-24T07:58:28Z","cross_cats_sorted":["cs.LG","cs.RO"],"title_canon_sha256":"832214f58c191fd57dcd93c92950a2ec9a70b211531945e5c6523d95c5eee61c","abstract_canon_sha256":"b0830afcdbabfa779360f2802948361062c627e983f264e7b0144840c1f067da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:59.184572Z","signature_b64":"89x4yl24wYZszLI6b7+yaUCofcNXKBihot+pGi7EdaBmMWNPCMQ9XsG/O1vD7+5rqTOdwtHkwcd05aC/jSAJBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7273bb4619981c3a6abf3104de739e713001df26874a605315e572d39c80e2d7","last_reissued_at":"2026-07-05T10:21:59.184039Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:59.184039Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViViDex: Learning Vision-based Dexterous Manipulation from Human Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Etienne Arlaud, Ivan Laptev, Shizhe Chen, Zerui Chen","submitted_at":"2024-04-24T07:58:28Z","abstract_excerpt":"In this work, we aim to learn a unified vision-based policy for multi-fingered robot hands to manipulate a variety of objects in diverse poses. Though prior work has shown benefits of using human videos for policy learning, performance gains have been limited by the noise in estimated trajectories. Moreover, reliance on privileged object information such as ground-truth object states further limits the applicability in realistic scenarios. To address these limitations, we propose a new framework ViViDex to improve vision-based policy learning from human videos. It first uses reinforcement lear"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.15709","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.15709/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.15709","created_at":"2026-07-05T10:21:59.184094+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.15709v3","created_at":"2026-07-05T10:21:59.184094+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.15709","created_at":"2026-07-05T10:21:59.184094+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJZ3WRQZTAOD","created_at":"2026-07-05T10:21:59.184094+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJZ3WRQZTAODU2V7","created_at":"2026-07-05T10:21:59.184094+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJZ3WRQZ","created_at":"2026-07-05T10:21:59.184094+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27475","citing_title":"Support-Constrained RL Enables Real-World Policy Improvement without Real-World Experience","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05925","citing_title":"DexSynRefine: Synthesizing and Refining Human-Object Interaction Motion for Physically Feasible Dexterous Robot Actions","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21723","citing_title":"VLBiMan: Vision-Language Anchored One-Shot Demonstration Enables Generalizable Bimanual Robotic Manipulation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05925","citing_title":"DexSynRefine: Synthesizing and Refining Human-Object Interaction Motion for Physically Feasible Dexterous Robot Actions","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE","json":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE.json","graph_json":"https://pith.science/api/pith-number/OJZ3WRQZTAODU2V7GECN4446OE/graph.json","events_json":"https://pith.science/api/pith-number/OJZ3WRQZTAODU2V7GECN4446OE/events.json","paper":"https://pith.science/paper/OJZ3WRQZ"},"agent_actions":{"view_html":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE","download_json":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE.json","view_paper":"https://pith.science/paper/OJZ3WRQZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.15709&json=true","fetch_graph":"https://pith.science/api/pith-number/OJZ3WRQZTAODU2V7GECN4446OE/graph.json","fetch_events":"https://pith.science/api/pith-number/OJZ3WRQZTAODU2V7GECN4446OE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE/action/storage_attestation","attest_author":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE/action/author_attestation","sign_citation":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE/action/citation_signature","submit_replication":"https://pith.science/pith/OJZ3WRQZTAODU2V7GECN4446OE/action/replication_record"}},"created_at":"2026-07-05T10:21:59.184094+00:00","updated_at":"2026-07-05T10:21:59.184094+00:00"}