{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K3Z2D6AHO7QCEVY432RLVYQZ6S","short_pith_number":"pith:K3Z2D6AH","schema_version":"1.0","canonical_sha256":"56f3a1f80777e022571cdea2bae219f4aed060715ad10ffeed8d938524f824ec","source":{"kind":"arxiv","id":"2503.00778","version":1},"attestation_state":"computed","paper":{"title":"AffordGrasp: In-Context Affordance Reasoning for Open-Vocabulary Task-Oriented Grasping in Clutter","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Jianlong Wu, Pengwei Wang, Shanghang Zhang, Shuaike Zhang, Xiaoshuai Hao, Yingbo Tang, Zhongyuan Wang","submitted_at":"2025-03-02T08:04:44Z","abstract_excerpt":"Inferring the affordance of an object and grasping it in a task-oriented manner is crucial for robots to successfully complete manipulation tasks. Affordance indicates where and how to grasp an object by taking its functionality into account, serving as the foundation for effective task-oriented grasping. However, current task-oriented methods often depend on extensive training data that is confined to specific tasks and objects, making it difficult to generalize to novel objects and complex scenes. In this paper, we introduce AffordGrasp, a novel open-vocabulary grasping framework that levera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00778","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-03-02T08:04:44Z","cross_cats_sorted":[],"title_canon_sha256":"ee4a9fa9898f978a6adc06591f6de0ccf5453279e5015ac21543b2f8d0eb9471","abstract_canon_sha256":"9767575369c90e709be170ec3901f3656688c13668d9db1a590f41a5d225f248"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:29.324849Z","signature_b64":"6ZMzXMNPk3KjTdYHHtH3YnCqceAGvQYBIgPmnaYIzvy04wgenkImmh6fu9cRhpLZ1FUf9PlQzGKD09y24lM7CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56f3a1f80777e022571cdea2bae219f4aed060715ad10ffeed8d938524f824ec","last_reissued_at":"2026-07-05T10:22:29.324242Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:29.324242Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AffordGrasp: In-Context Affordance Reasoning for Open-Vocabulary Task-Oriented Grasping in Clutter","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Jianlong Wu, Pengwei Wang, Shanghang Zhang, Shuaike Zhang, Xiaoshuai Hao, Yingbo Tang, Zhongyuan Wang","submitted_at":"2025-03-02T08:04:44Z","abstract_excerpt":"Inferring the affordance of an object and grasping it in a task-oriented manner is crucial for robots to successfully complete manipulation tasks. Affordance indicates where and how to grasp an object by taking its functionality into account, serving as the foundation for effective task-oriented grasping. However, current task-oriented methods often depend on extensive training data that is confined to specific tasks and objects, making it difficult to generalize to novel objects and complex scenes. In this paper, we introduce AffordGrasp, a novel open-vocabulary grasping framework that levera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00778","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00778","created_at":"2026-07-05T10:22:29.324326+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00778v1","created_at":"2026-07-05T10:22:29.324326+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00778","created_at":"2026-07-05T10:22:29.324326+00:00"},{"alias_kind":"pith_short_12","alias_value":"K3Z2D6AHO7QC","created_at":"2026-07-05T10:22:29.324326+00:00"},{"alias_kind":"pith_short_16","alias_value":"K3Z2D6AHO7QCEVY4","created_at":"2026-07-05T10:22:29.324326+00:00"},{"alias_kind":"pith_short_8","alias_value":"K3Z2D6AH","created_at":"2026-07-05T10:22:29.324326+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09798","citing_title":"SynManDex: Synthesizing Human-like Dexterous Grasps from Synthetic Human Pre-Grasps","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":206,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13451","citing_title":"MapNav: A Novel Memory Representation via Annotated Semantic Maps for Vision-and-Language Navigation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22003","citing_title":"VP-VLA: Visual Prompting as an Interface for Vision-Language-Action Models","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S","json":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S.json","graph_json":"https://pith.science/api/pith-number/K3Z2D6AHO7QCEVY432RLVYQZ6S/graph.json","events_json":"https://pith.science/api/pith-number/K3Z2D6AHO7QCEVY432RLVYQZ6S/events.json","paper":"https://pith.science/paper/K3Z2D6AH"},"agent_actions":{"view_html":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S","download_json":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S.json","view_paper":"https://pith.science/paper/K3Z2D6AH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00778&json=true","fetch_graph":"https://pith.science/api/pith-number/K3Z2D6AHO7QCEVY432RLVYQZ6S/graph.json","fetch_events":"https://pith.science/api/pith-number/K3Z2D6AHO7QCEVY432RLVYQZ6S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S/action/storage_attestation","attest_author":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S/action/author_attestation","sign_citation":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S/action/citation_signature","submit_replication":"https://pith.science/pith/K3Z2D6AHO7QCEVY432RLVYQZ6S/action/replication_record"}},"created_at":"2026-07-05T10:22:29.324326+00:00","updated_at":"2026-07-05T10:22:29.324326+00:00"}