{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:SEG56LOSNWZH6CUM7M6OYNOVAF","short_pith_number":"pith:SEG56LOS","schema_version":"1.0","canonical_sha256":"910ddf2dd26db27f0a8cfb3cec35d5016e445238650c59c09e5355bd3b7e1c14","source":{"kind":"arxiv","id":"1912.04443","version":3},"attestation_state":"computed","paper":{"title":"AVID: Learning Multi-Stage Tasks via Pixel-Level Translation of Human Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Laura Smith, Marvin Zhang, Nikita Dhawan, Pieter Abbeel, Sergey Levine","submitted_at":"2019-12-10T01:36:18Z","abstract_excerpt":"Robotic reinforcement learning (RL) holds the promise of enabling robots to learn complex behaviors through experience. However, realizing this promise for long-horizon tasks in the real world requires mechanisms to reduce human burden in terms of defining the task and scaffolding the learning process. In this paper, we study how these challenges can be alleviated with an automated robotic learning framework, in which multi-stage tasks are defined simply by providing videos of a human demonstrator and then learned autonomously by the robot from raw image observations. A central challenge in im"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.04443","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-12-10T01:36:18Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"1cd9fabd37c10354bf296f73ceb5459bca07480253e8e708cc7750afeec13481","abstract_canon_sha256":"8702d58898fdcf7003bca3b2a283957a23dd298394f92a87eb529c5590e06a9c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:11:43.559422Z","signature_b64":"4H4h2btWQNzqnM706N0hteu5o4QMxHCGAXAyulj/tOp0E7rM6vFEYTzIutz3JLV0gecmai6jQL8QosAQw6L6Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"910ddf2dd26db27f0a8cfb3cec35d5016e445238650c59c09e5355bd3b7e1c14","last_reissued_at":"2026-07-05T01:11:43.558929Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:11:43.558929Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AVID: Learning Multi-Stage Tasks via Pixel-Level Translation of Human Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Laura Smith, Marvin Zhang, Nikita Dhawan, Pieter Abbeel, Sergey Levine","submitted_at":"2019-12-10T01:36:18Z","abstract_excerpt":"Robotic reinforcement learning (RL) holds the promise of enabling robots to learn complex behaviors through experience. However, realizing this promise for long-horizon tasks in the real world requires mechanisms to reduce human burden in terms of defining the task and scaffolding the learning process. In this paper, we study how these challenges can be alleviated with an automated robotic learning framework, in which multi-stage tasks are defined simply by providing videos of a human demonstrator and then learned autonomously by the robot from raw image observations. A central challenge in im"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.04443","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.04443/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.04443","created_at":"2026-07-05T01:11:43.558995+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.04443v3","created_at":"2026-07-05T01:11:43.558995+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.04443","created_at":"2026-07-05T01:11:43.558995+00:00"},{"alias_kind":"pith_short_12","alias_value":"SEG56LOSNWZH","created_at":"2026-07-05T01:11:43.558995+00:00"},{"alias_kind":"pith_short_16","alias_value":"SEG56LOSNWZH6CUM","created_at":"2026-07-05T01:11:43.558995+00:00"},{"alias_kind":"pith_short_8","alias_value":"SEG56LOS","created_at":"2026-07-05T01:11:43.558995+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03201","citing_title":"Reinforcement Learning from Cross-domain Videos with Video Prediction Model","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00990","citing_title":"Robotic Manipulation by Imitating Generated Videos Without Physical Demonstrations","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02117","citing_title":"Mobile ALOHA: Learning Bimanual Mobile Manipulation with Low-Cost Whole-Body Teleoperation","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04974","citing_title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2310.08864","citing_title":"Open X-Embodiment: Robotic Learning Datasets and RT-X Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07607","citing_title":"EgoVerse: An Egocentric Human Dataset for Robot Learning from Around the World","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF","json":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF.json","graph_json":"https://pith.science/api/pith-number/SEG56LOSNWZH6CUM7M6OYNOVAF/graph.json","events_json":"https://pith.science/api/pith-number/SEG56LOSNWZH6CUM7M6OYNOVAF/events.json","paper":"https://pith.science/paper/SEG56LOS"},"agent_actions":{"view_html":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF","download_json":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF.json","view_paper":"https://pith.science/paper/SEG56LOS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.04443&json=true","fetch_graph":"https://pith.science/api/pith-number/SEG56LOSNWZH6CUM7M6OYNOVAF/graph.json","fetch_events":"https://pith.science/api/pith-number/SEG56LOSNWZH6CUM7M6OYNOVAF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF/action/storage_attestation","attest_author":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF/action/author_attestation","sign_citation":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF/action/citation_signature","submit_replication":"https://pith.science/pith/SEG56LOSNWZH6CUM7M6OYNOVAF/action/replication_record"}},"created_at":"2026-07-05T01:11:43.558995+00:00","updated_at":"2026-07-05T01:11:43.558995+00:00"}