{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NYHI24R6QA2GTMKACZ4DUZ2GLN","short_pith_number":"pith:NYHI24R6","schema_version":"1.0","canonical_sha256":"6e0e8d723e803469b14016783a67465b6ba0a9ef2bf008ff2dcf8967eace1ab4","source":{"kind":"arxiv","id":"2501.03968","version":2},"attestation_state":"computed","paper":{"title":"VLM-driven Behavior Tree for Context-aware Task Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.HC"],"primary_cat":"cs.RO","authors_text":"Atsushi Kanehira, Jun Takamatsu, Katsushi Ikeuchi, Kazuhiro Sasabuchi, Naoki Wake","submitted_at":"2025-01-07T18:06:27Z","abstract_excerpt":"The use of Large Language Models (LLMs) for generating Behavior Trees (BTs) has recently gained attention in the robotics community, yet remains in its early stages of development. In this paper, we propose a novel framework that leverages Vision-Language Models (VLMs) to interactively generate and edit BTs that address visual conditions, enabling context-aware robot operations in visually complex environments. A key feature of our approach lies in the conditional control through self-prompted visual conditions. Specifically, the VLM generates BTs with visual condition nodes, where conditions "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03968","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-01-07T18:06:27Z","cross_cats_sorted":["cs.AI","cs.CV","cs.HC"],"title_canon_sha256":"d1f6be69aabe7d31722a5a8f8ffbf67812874063c66efc29e5f1a6a85b1a72e9","abstract_canon_sha256":"81f896362e9d91a94b2505d82200a2e2c40da3157896e009a47f1d477beacca6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:22.792117Z","signature_b64":"woy0IcNHe0w9ES2sD/Xh8l0purQpJVt5D7maid2czWtVssUbNVkXUo8emEJJPE1d69LJ/v2fxU/AxayZe0KuBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e0e8d723e803469b14016783a67465b6ba0a9ef2bf008ff2dcf8967eace1ab4","last_reissued_at":"2026-07-05T09:59:22.791659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:22.791659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VLM-driven Behavior Tree for Context-aware Task Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.HC"],"primary_cat":"cs.RO","authors_text":"Atsushi Kanehira, Jun Takamatsu, Katsushi Ikeuchi, Kazuhiro Sasabuchi, Naoki Wake","submitted_at":"2025-01-07T18:06:27Z","abstract_excerpt":"The use of Large Language Models (LLMs) for generating Behavior Trees (BTs) has recently gained attention in the robotics community, yet remains in its early stages of development. In this paper, we propose a novel framework that leverages Vision-Language Models (VLMs) to interactively generate and edit BTs that address visual conditions, enabling context-aware robot operations in visually complex environments. A key feature of our approach lies in the conditional control through self-prompted visual conditions. Specifically, the VLM generates BTs with visual condition nodes, where conditions "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03968","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03968","created_at":"2026-07-05T09:59:22.791715+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03968v2","created_at":"2026-07-05T09:59:22.791715+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03968","created_at":"2026-07-05T09:59:22.791715+00:00"},{"alias_kind":"pith_short_12","alias_value":"NYHI24R6QA2G","created_at":"2026-07-05T09:59:22.791715+00:00"},{"alias_kind":"pith_short_16","alias_value":"NYHI24R6QA2GTMKA","created_at":"2026-07-05T09:59:22.791715+00:00"},{"alias_kind":"pith_short_8","alias_value":"NYHI24R6","created_at":"2026-07-05T09:59:22.791715+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02812","citing_title":"Learning Structured Robot Policies from Vision-Language Models via Synthetic Neuro-Symbolic Supervision","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02812","citing_title":"Learning Structured Robot Policies from Vision-Language Models via Synthetic Neuro-Symbolic Supervision","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN","json":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN.json","graph_json":"https://pith.science/api/pith-number/NYHI24R6QA2GTMKACZ4DUZ2GLN/graph.json","events_json":"https://pith.science/api/pith-number/NYHI24R6QA2GTMKACZ4DUZ2GLN/events.json","paper":"https://pith.science/paper/NYHI24R6"},"agent_actions":{"view_html":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN","download_json":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN.json","view_paper":"https://pith.science/paper/NYHI24R6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03968&json=true","fetch_graph":"https://pith.science/api/pith-number/NYHI24R6QA2GTMKACZ4DUZ2GLN/graph.json","fetch_events":"https://pith.science/api/pith-number/NYHI24R6QA2GTMKACZ4DUZ2GLN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN/action/storage_attestation","attest_author":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN/action/author_attestation","sign_citation":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN/action/citation_signature","submit_replication":"https://pith.science/pith/NYHI24R6QA2GTMKACZ4DUZ2GLN/action/replication_record"}},"created_at":"2026-07-05T09:59:22.791715+00:00","updated_at":"2026-07-05T09:59:22.791715+00:00"}