{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MVH5YB3FP6RQK76JOAT3IFFF5N","short_pith_number":"pith:MVH5YB3F","schema_version":"1.0","canonical_sha256":"654fdc07657fa3057fc97027b414a5eb79a8538e4b6cef2bf23bc5719c36db95","source":{"kind":"arxiv","id":"2308.12537","version":1},"attestation_state":"computed","paper":{"title":"HuBo-VLM: Unified Vision-Language Model designed for HUman roBOt interaction tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Hang Ji, Junbo Chen, Weikun Zhang, Xin Zhan, Xufeng Huang, Zichao Dong","submitted_at":"2023-08-24T03:47:27Z","abstract_excerpt":"Human robot interaction is an exciting task, which aimed to guide robots following instructions from human. Since huge gap lies between human natural language and machine codes, end to end human robot interaction models is fair challenging. Further, visual information receiving from sensors of robot is also a hard language for robot to perceive. In this work, HuBo-VLM is proposed to tackle perception tasks associated with human robot interaction including object detection and visual grounding by a unified transformer based vision language model. Extensive experiments on the Talk2Car benchmark "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.12537","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-08-24T03:47:27Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"0a68e0ccdd9a004bf7128b75cb3a7da2cea763f869e145a6c9a1d0c4a6715b85","abstract_canon_sha256":"9e477f38dd40c09b51350b77421600bef197725307449036215d368e6c163bdb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:44:14.129714Z","signature_b64":"aFMaeLo/4ZGLV+v6EVKQWRv/fwlEQC2q6+/Rl1U+5gV0624HOod+JgpI+4kKuaDAKD1DPjQrxgZreFcriKLPDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"654fdc07657fa3057fc97027b414a5eb79a8538e4b6cef2bf23bc5719c36db95","last_reissued_at":"2026-07-05T06:44:14.129305Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:44:14.129305Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HuBo-VLM: Unified Vision-Language Model designed for HUman roBOt interaction tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Hang Ji, Junbo Chen, Weikun Zhang, Xin Zhan, Xufeng Huang, Zichao Dong","submitted_at":"2023-08-24T03:47:27Z","abstract_excerpt":"Human robot interaction is an exciting task, which aimed to guide robots following instructions from human. Since huge gap lies between human natural language and machine codes, end to end human robot interaction models is fair challenging. Further, visual information receiving from sensors of robot is also a hard language for robot to perceive. In this work, HuBo-VLM is proposed to tackle perception tasks associated with human robot interaction including object detection and visual grounding by a unified transformer based vision language model. Extensive experiments on the Talk2Car benchmark "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.12537","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.12537/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.12537","created_at":"2026-07-05T06:44:14.129371+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.12537v1","created_at":"2026-07-05T06:44:14.129371+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.12537","created_at":"2026-07-05T06:44:14.129371+00:00"},{"alias_kind":"pith_short_12","alias_value":"MVH5YB3FP6RQ","created_at":"2026-07-05T06:44:14.129371+00:00"},{"alias_kind":"pith_short_16","alias_value":"MVH5YB3FP6RQK76J","created_at":"2026-07-05T06:44:14.129371+00:00"},{"alias_kind":"pith_short_8","alias_value":"MVH5YB3F","created_at":"2026-07-05T06:44:14.129371+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.08757","citing_title":"SocialNav-SUB: Benchmarking VLMs for Scene Understanding in Social Robot Navigation","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N","json":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N.json","graph_json":"https://pith.science/api/pith-number/MVH5YB3FP6RQK76JOAT3IFFF5N/graph.json","events_json":"https://pith.science/api/pith-number/MVH5YB3FP6RQK76JOAT3IFFF5N/events.json","paper":"https://pith.science/paper/MVH5YB3F"},"agent_actions":{"view_html":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N","download_json":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N.json","view_paper":"https://pith.science/paper/MVH5YB3F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.12537&json=true","fetch_graph":"https://pith.science/api/pith-number/MVH5YB3FP6RQK76JOAT3IFFF5N/graph.json","fetch_events":"https://pith.science/api/pith-number/MVH5YB3FP6RQK76JOAT3IFFF5N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N/action/storage_attestation","attest_author":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N/action/author_attestation","sign_citation":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N/action/citation_signature","submit_replication":"https://pith.science/pith/MVH5YB3FP6RQK76JOAT3IFFF5N/action/replication_record"}},"created_at":"2026-07-05T06:44:14.129371+00:00","updated_at":"2026-07-05T06:44:14.129371+00:00"}