{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MZ3U4SHOEXXLWLD6AQDMVNKYF2","short_pith_number":"pith:MZ3U4SHO","schema_version":"1.0","canonical_sha256":"66774e48ee25eebb2c7e0406cab5582ea36e6c8a08e758f70bd42d1cbf0f53d9","source":{"kind":"arxiv","id":"2411.02704","version":1},"attestation_state":"computed","paper":{"title":"RT-Affordance: Affordances are Versatile Intermediate Representations for Robot Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Danny Driess, Dorsa Sadigh, Laura Smith, Sean Kirmani, Soroush Nasiriany, Ted Xiao, Tianli Ding, Yuke Zhu","submitted_at":"2024-11-05T01:02:51Z","abstract_excerpt":"We explore how intermediate policy representations can facilitate generalization by providing guidance on how to perform manipulation tasks. Existing representations such as language, goal images, and trajectory sketches have been shown to be helpful, but these representations either do not provide enough context or provide over-specified context that yields less robust policies. We propose conditioning policies on affordances, which capture the pose of the robot at key stages of the task. Affordances offer expressive yet lightweight abstractions, are easy for users to specify, and facilitate "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.02704","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-11-05T01:02:51Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"b9ce30de1914e310196cf0cc8d96e0106ca2b64b6f35dbfe713f80fc6562452c","abstract_canon_sha256":"251dcb1f037caa386cefb61420aaa40436fc0cf513eb3c3211e454362a847aca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:16.622271Z","signature_b64":"8/hMXwUIWvr9W3D+iLdf9GPrhXBJbH9Ws6eRjWc9n39y/ZUvB1EV4alB2jVw9lbIloyXdzctNlW9Z6ogF/DxAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66774e48ee25eebb2c7e0406cab5582ea36e6c8a08e758f70bd42d1cbf0f53d9","last_reissued_at":"2026-07-05T09:31:16.621791Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:16.621791Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RT-Affordance: Affordances are Versatile Intermediate Representations for Robot Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Danny Driess, Dorsa Sadigh, Laura Smith, Sean Kirmani, Soroush Nasiriany, Ted Xiao, Tianli Ding, Yuke Zhu","submitted_at":"2024-11-05T01:02:51Z","abstract_excerpt":"We explore how intermediate policy representations can facilitate generalization by providing guidance on how to perform manipulation tasks. Existing representations such as language, goal images, and trajectory sketches have been shown to be helpful, but these representations either do not provide enough context or provide over-specified context that yields less robust policies. We propose conditioning policies on affordances, which capture the pose of the robot at key stages of the task. Affordances offer expressive yet lightweight abstractions, are easy for users to specify, and facilitate "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.02704","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.02704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.02704","created_at":"2026-07-05T09:31:16.621844+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.02704v1","created_at":"2026-07-05T09:31:16.621844+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.02704","created_at":"2026-07-05T09:31:16.621844+00:00"},{"alias_kind":"pith_short_12","alias_value":"MZ3U4SHOEXXL","created_at":"2026-07-05T09:31:16.621844+00:00"},{"alias_kind":"pith_short_16","alias_value":"MZ3U4SHOEXXLWLD6","created_at":"2026-07-05T09:31:16.621844+00:00"},{"alias_kind":"pith_short_8","alias_value":"MZ3U4SHO","created_at":"2026-07-05T09:31:16.621844+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29089","citing_title":"TAP-VLA: Tactile Annotation Prompting for Vision Language Action Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02239","citing_title":"LACY: A Vision-Language Model-based Language-Action Cycle for Self-Improving Robotic Manipulation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13998","citing_title":"Embodied-R1: Reinforced Embodied Reasoning for General Robotic Manipulation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21723","citing_title":"VLBiMan: Vision-Language Anchored One-Shot Demonstration Enables Generalizable Bimanual Robotic Manipulation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13778","citing_title":"InternVLA-M1: A Spatially Guided Vision-Language-Action Framework for Generalist Robot Policy","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03637","citing_title":"Bridging the Embodiment Gap: Disentangled Cross-Embodiment Video Editing","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22911","citing_title":"RecoverFormer: End-to-End Contact-Aware Recovery for Humanoid Robots","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00663","citing_title":"Affordance Agent Harness: Verification-Gated Skill Orchestration","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00663","citing_title":"Affordance Agent Harness: Verification-Gated Skill Orchestration","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2","json":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2.json","graph_json":"https://pith.science/api/pith-number/MZ3U4SHOEXXLWLD6AQDMVNKYF2/graph.json","events_json":"https://pith.science/api/pith-number/MZ3U4SHOEXXLWLD6AQDMVNKYF2/events.json","paper":"https://pith.science/paper/MZ3U4SHO"},"agent_actions":{"view_html":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2","download_json":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2.json","view_paper":"https://pith.science/paper/MZ3U4SHO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.02704&json=true","fetch_graph":"https://pith.science/api/pith-number/MZ3U4SHOEXXLWLD6AQDMVNKYF2/graph.json","fetch_events":"https://pith.science/api/pith-number/MZ3U4SHOEXXLWLD6AQDMVNKYF2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2/action/storage_attestation","attest_author":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2/action/author_attestation","sign_citation":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2/action/citation_signature","submit_replication":"https://pith.science/pith/MZ3U4SHOEXXLWLD6AQDMVNKYF2/action/replication_record"}},"created_at":"2026-07-05T09:31:16.621844+00:00","updated_at":"2026-07-05T09:31:16.621844+00:00"}