{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FP7IYUKKJHUS7KAYDI43LF53ED","short_pith_number":"pith:FP7IYUKK","schema_version":"1.0","canonical_sha256":"2bfe8c514a49e92fa8181a39b597bb20c71fae2c62bb8b436ab2e8cd7a0a54f9","source":{"kind":"arxiv","id":"2406.07549","version":2},"attestation_state":"computed","paper":{"title":"A3VLM: Actionable Articulation-Aware Vision Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Abdeslam Boularias, Hao Dong, Haonan Chang, Hongsheng Li, Peng Gao, Siyuan Huang, Yimeng Zhu, Yuhan Liu","submitted_at":"2024-06-11T17:59:55Z","abstract_excerpt":"Vision Language Models (VLMs) have received significant attention in recent years in the robotics community. VLMs are shown to be able to perform complex visual reasoning and scene understanding tasks, which makes them regarded as a potential universal solution for general robotics problems such as manipulation and navigation. However, previous VLMs for robotics such as RT-1, RT-2, and ManipLLM have focused on directly learning robot-centric actions. Such approaches require collecting a significant amount of robot interaction data, which is extremely costly in the real world. Thus, we propose "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07549","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-06-11T17:59:55Z","cross_cats_sorted":[],"title_canon_sha256":"d6e6a1ecfd505ce4e93175879bb51f1e5f7c77ab7694756e1265855c91fffa0e","abstract_canon_sha256":"f8453a2abda3d9007e655780a4d6fe52dd1268eff1df8d903bf12fbdd101583a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:17.052550Z","signature_b64":"P6eoxNNXZFaQB+IoxR903Lu0qabFs5Jb7ovg7OBBU0vCEzf74th5dMD5erZhoTG3d2jzMz9t2TBGot7JEUYoCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2bfe8c514a49e92fa8181a39b597bb20c71fae2c62bb8b436ab2e8cd7a0a54f9","last_reissued_at":"2026-07-05T08:31:17.052018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:17.052018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A3VLM: Actionable Articulation-Aware Vision Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Abdeslam Boularias, Hao Dong, Haonan Chang, Hongsheng Li, Peng Gao, Siyuan Huang, Yimeng Zhu, Yuhan Liu","submitted_at":"2024-06-11T17:59:55Z","abstract_excerpt":"Vision Language Models (VLMs) have received significant attention in recent years in the robotics community. VLMs are shown to be able to perform complex visual reasoning and scene understanding tasks, which makes them regarded as a potential universal solution for general robotics problems such as manipulation and navigation. However, previous VLMs for robotics such as RT-1, RT-2, and ManipLLM have focused on directly learning robot-centric actions. Such approaches require collecting a significant amount of robot interaction data, which is extremely costly in the real world. Thus, we propose "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07549","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07549/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07549","created_at":"2026-07-05T08:31:17.052090+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07549v2","created_at":"2026-07-05T08:31:17.052090+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07549","created_at":"2026-07-05T08:31:17.052090+00:00"},{"alias_kind":"pith_short_12","alias_value":"FP7IYUKKJHUS","created_at":"2026-07-05T08:31:17.052090+00:00"},{"alias_kind":"pith_short_16","alias_value":"FP7IYUKKJHUS7KAY","created_at":"2026-07-05T08:31:17.052090+00:00"},{"alias_kind":"pith_short_8","alias_value":"FP7IYUKK","created_at":"2026-07-05T08:31:17.052090+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30608","citing_title":"UnfoldArt: Zero-Shot Recovery of Full Articulated 3D Objects from Text or Image","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30608","citing_title":"UnfoldArt: Zero-Shot Recovery of Full Articulated 3D Objects from Text or Image","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29774","citing_title":"Analytic Concept-Centric Memory for Agentic Embodied Manipulation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22183","citing_title":"Action with Visual Primitives","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16582","citing_title":"ArtMesh: Part-Aware Articulated Mesh Fields with Motion-Consistent Dynamics","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02812","citing_title":"Learning Structured Robot Policies from Vision-Language Models via Synthetic Neuro-Symbolic Supervision","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13998","citing_title":"Embodied-R1: Reinforced Embodied Reasoning for General Robotic Manipulation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2601.07060","citing_title":"PALM: Progress-Aware Policy Learning via Affordance Reasoning for Long-Horizon Robotic Manipulation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2409.01652","citing_title":"ReKep: Spatio-Temporal Reasoning of Relational Keypoint Constraints for Robotic Manipulation","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02812","citing_title":"Learning Structured Robot Policies from Vision-Language Models via Synthetic Neuro-Symbolic Supervision","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25781","citing_title":"Sketch2Arti: Sketch-based Articulation Modeling of CAD Objects","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED","json":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED.json","graph_json":"https://pith.science/api/pith-number/FP7IYUKKJHUS7KAYDI43LF53ED/graph.json","events_json":"https://pith.science/api/pith-number/FP7IYUKKJHUS7KAYDI43LF53ED/events.json","paper":"https://pith.science/paper/FP7IYUKK"},"agent_actions":{"view_html":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED","download_json":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED.json","view_paper":"https://pith.science/paper/FP7IYUKK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07549&json=true","fetch_graph":"https://pith.science/api/pith-number/FP7IYUKKJHUS7KAYDI43LF53ED/graph.json","fetch_events":"https://pith.science/api/pith-number/FP7IYUKKJHUS7KAYDI43LF53ED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED/action/storage_attestation","attest_author":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED/action/author_attestation","sign_citation":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED/action/citation_signature","submit_replication":"https://pith.science/pith/FP7IYUKKJHUS7KAYDI43LF53ED/action/replication_record"}},"created_at":"2026-07-05T08:31:17.052090+00:00","updated_at":"2026-07-05T08:31:17.052090+00:00"}