{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XQAFQHDZ3Z5UC2V2P4WYY2ZHB6","short_pith_number":"pith:XQAFQHDZ","schema_version":"1.0","canonical_sha256":"bc00581c79de7b416aba7f2d8c6b270fb24e429a008b9042b92ad55d962b2193","source":{"kind":"arxiv","id":"2407.21762","version":1},"attestation_state":"computed","paper":{"title":"ReplanVLM: Replanning Robotic Tasks with Visual Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Aoran Mei, Guo-Niu Zhu, Huaxiang Zhang, Zhongxue Gan","submitted_at":"2024-07-31T17:31:01Z","abstract_excerpt":"Large language models (LLMs) have gained increasing popularity in robotic task planning due to their exceptional abilities in text analytics and generation, as well as their broad knowledge of the world. However, they fall short in decoding visual cues. LLMs have limited direct perception of the world, which leads to a deficient grasp of the current state of the world. By contrast, the emergence of visual language models (VLMs) fills this gap by integrating visual perception modules, which can enhance the autonomy of robotic task planning. Despite these advancements, VLMs still face challenges"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.21762","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-07-31T17:31:01Z","cross_cats_sorted":[],"title_canon_sha256":"27c8f42f6e74829201c5b8ca9e3df648cd359b5a2bd1dcf5978a9843e91ede46","abstract_canon_sha256":"76287690f4331083236c2b2eb84c531df6e842f48c4b2486103fab7d131217df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:43.321677Z","signature_b64":"7YItHJlp7XFAG1Evmum43zsh5qy1sGVfJJZQ7mbzg4CJorufTO2ldnQ/ZZaI4b2fnnNuT58YVWrR+Lgp71MHDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bc00581c79de7b416aba7f2d8c6b270fb24e429a008b9042b92ad55d962b2193","last_reissued_at":"2026-07-05T08:50:43.321252Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:43.321252Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReplanVLM: Replanning Robotic Tasks with Visual Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Aoran Mei, Guo-Niu Zhu, Huaxiang Zhang, Zhongxue Gan","submitted_at":"2024-07-31T17:31:01Z","abstract_excerpt":"Large language models (LLMs) have gained increasing popularity in robotic task planning due to their exceptional abilities in text analytics and generation, as well as their broad knowledge of the world. However, they fall short in decoding visual cues. LLMs have limited direct perception of the world, which leads to a deficient grasp of the current state of the world. By contrast, the emergence of visual language models (VLMs) fills this gap by integrating visual perception modules, which can enhance the autonomy of robotic task planning. Despite these advancements, VLMs still face challenges"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.21762","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.21762/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.21762","created_at":"2026-07-05T08:50:43.321324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.21762v1","created_at":"2026-07-05T08:50:43.321324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.21762","created_at":"2026-07-05T08:50:43.321324+00:00"},{"alias_kind":"pith_short_12","alias_value":"XQAFQHDZ3Z5U","created_at":"2026-07-05T08:50:43.321324+00:00"},{"alias_kind":"pith_short_16","alias_value":"XQAFQHDZ3Z5UC2V2","created_at":"2026-07-05T08:50:43.321324+00:00"},{"alias_kind":"pith_short_8","alias_value":"XQAFQHDZ","created_at":"2026-07-05T08:50:43.321324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07723","citing_title":"VoLo: A Physical Orchestrator for Open-Vocabulary Long-Horizon Manipulation","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6","json":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6.json","graph_json":"https://pith.science/api/pith-number/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/graph.json","events_json":"https://pith.science/api/pith-number/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/events.json","paper":"https://pith.science/paper/XQAFQHDZ"},"agent_actions":{"view_html":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6","download_json":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6.json","view_paper":"https://pith.science/paper/XQAFQHDZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.21762&json=true","fetch_graph":"https://pith.science/api/pith-number/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/graph.json","fetch_events":"https://pith.science/api/pith-number/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/action/storage_attestation","attest_author":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/action/author_attestation","sign_citation":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/action/citation_signature","submit_replication":"https://pith.science/pith/XQAFQHDZ3Z5UC2V2P4WYY2ZHB6/action/replication_record"}},"created_at":"2026-07-05T08:50:43.321324+00:00","updated_at":"2026-07-05T08:50:43.321324+00:00"}