{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RE225CSKBWJN37HHKZAOIYKSHY","short_pith_number":"pith:RE225CSK","schema_version":"1.0","canonical_sha256":"8935ae8a4a0d92ddfce75640e461523e07c37bf226d56655df606bd1b6eaecb9","source":{"kind":"arxiv","id":"2304.08488","version":1},"attestation_state":"computed","paper":{"title":"Affordances from Human Videos as a Versatile Representation for Robotics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.NE"],"primary_cat":"cs.RO","authors_text":"Deepak Pathak, Lili Chen, Russell Mendonca, Shikhar Bahl, Unnat Jain","submitted_at":"2023-04-17T17:59:34Z","abstract_excerpt":"Building a robot that can understand and learn to interact by watching humans has inspired several vision problems. However, despite some successful results on static datasets, it remains unclear how current models can be used on a robot directly. In this paper, we aim to bridge this gap by leveraging videos of human interactions in an environment centric manner. Utilizing internet videos of human behavior, we train a visual affordance model that estimates where and how in the scene a human is likely to interact. The structure of these behavioral affordances directly enables the robot to perfo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.08488","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-04-17T17:59:34Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG","cs.NE"],"title_canon_sha256":"360f82697d99dea6556a8e97a1c64c90f7d980da880379f69111f93b64e78a31","abstract_canon_sha256":"43e4335d87bb638a670f6d2b3e76e845a829fe95ee9964a3f1a0df47de961d50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:01:53.018828Z","signature_b64":"z4MBWmitmoU+p9ZiQ1b/vWXls1/j/PBAcyaUeP1JgT15Ma/apcCnhLnRD5IM+L5byNNtS212svu9MLevjW2sAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8935ae8a4a0d92ddfce75640e461523e07c37bf226d56655df606bd1b6eaecb9","last_reissued_at":"2026-07-05T06:01:53.018362Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:01:53.018362Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Affordances from Human Videos as a Versatile Representation for Robotics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.NE"],"primary_cat":"cs.RO","authors_text":"Deepak Pathak, Lili Chen, Russell Mendonca, Shikhar Bahl, Unnat Jain","submitted_at":"2023-04-17T17:59:34Z","abstract_excerpt":"Building a robot that can understand and learn to interact by watching humans has inspired several vision problems. However, despite some successful results on static datasets, it remains unclear how current models can be used on a robot directly. In this paper, we aim to bridge this gap by leveraging videos of human interactions in an environment centric manner. Utilizing internet videos of human behavior, we train a visual affordance model that estimates where and how in the scene a human is likely to interact. The structure of these behavioral affordances directly enables the robot to perfo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.08488","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.08488/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.08488","created_at":"2026-07-05T06:01:53.018423+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.08488v1","created_at":"2026-07-05T06:01:53.018423+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.08488","created_at":"2026-07-05T06:01:53.018423+00:00"},{"alias_kind":"pith_short_12","alias_value":"RE225CSKBWJN","created_at":"2026-07-05T06:01:53.018423+00:00"},{"alias_kind":"pith_short_16","alias_value":"RE225CSKBWJN37HH","created_at":"2026-07-05T06:01:53.018423+00:00"},{"alias_kind":"pith_short_8","alias_value":"RE225CSK","created_at":"2026-07-05T06:01:53.018423+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06155","citing_title":"AffordanceVLA: A Vision-Language-Action Model Empowering Action Generation through Affordance-Aware Understanding","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05533","citing_title":"What Objects Enable, Not What They Are: Functional Latent Spaces for Affordance Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04974","citing_title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02792","citing_title":"Unified World Models: Coupling Video and Action Diffusion for Pretraining on Large Robotic Datasets","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10809","citing_title":"WARPED: Wrist-Aligned Rendering for Robot Policy Learning from Egocentric Human Demonstrations","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY","json":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY.json","graph_json":"https://pith.science/api/pith-number/RE225CSKBWJN37HHKZAOIYKSHY/graph.json","events_json":"https://pith.science/api/pith-number/RE225CSKBWJN37HHKZAOIYKSHY/events.json","paper":"https://pith.science/paper/RE225CSK"},"agent_actions":{"view_html":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY","download_json":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY.json","view_paper":"https://pith.science/paper/RE225CSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.08488&json=true","fetch_graph":"https://pith.science/api/pith-number/RE225CSKBWJN37HHKZAOIYKSHY/graph.json","fetch_events":"https://pith.science/api/pith-number/RE225CSKBWJN37HHKZAOIYKSHY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY/action/storage_attestation","attest_author":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY/action/author_attestation","sign_citation":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY/action/citation_signature","submit_replication":"https://pith.science/pith/RE225CSKBWJN37HHKZAOIYKSHY/action/replication_record"}},"created_at":"2026-07-05T06:01:53.018423+00:00","updated_at":"2026-07-05T06:01:53.018423+00:00"}