{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IOYHG2TWNZLJXXDOOZPWSZJHAX","short_pith_number":"pith:IOYHG2TW","schema_version":"1.0","canonical_sha256":"43b0736a766e569bdc6e765f69652705cb473a403906660d0adf1000034dacfc","source":{"kind":"arxiv","id":"2506.00123","version":1},"attestation_state":"computed","paper":{"title":"Visual Embodied Brain: Let Multimodal Large Language Models See, Think, and Control in Spaces","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Erfei Cui, Ganlin Yang, Gen Luo, Guanzhou Chen, Haonan Duan, Jifeng Dai, Jingbo Wang, Lewei Lu, Ronglei Tong, Rongrong Ji, Shenglong Ye, Tianyi Zhang, Wenhai Wang, Xizhou Zhu, Yu Qiao, Zhe Chen, Zhi Hou, Ziyang Gong","submitted_at":"2025-05-30T18:00:34Z","abstract_excerpt":"The remarkable progress of Multimodal Large Language Models (MLLMs) has attracted increasing attention to extend them to physical entities like legged robot. This typically requires MLLMs to not only grasp multimodal understanding abilities, but also integrate visual-spatial reasoning and physical interaction capabilities. Nevertheless,existing methods struggle to unify these capabilities due to their fundamental differences.In this paper, we present the Visual Embodied Brain (VeBrain), a unified framework for perception, reasoning, and control in real world. VeBrain reformulates robotic contr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00123","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-30T18:00:34Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"71f204e3a5670b147f579ebf31d9e384d79d3e56389d14bcffe21874e7c05742","abstract_canon_sha256":"03fc39b5571b198c37d0b9d7dcba5f22fc71d9bc15c4748d0294982154dd4ffb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:29.310022Z","signature_b64":"PV9yFzrQIvc8tQ2Wtrh+Fbc1THN+tIi/QBUuGr0hnAfWhmQlc8klB3WUw3ghR1klTTsgYTkBOrArR99ezVXyCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43b0736a766e569bdc6e765f69652705cb473a403906660d0adf1000034dacfc","last_reissued_at":"2026-07-05T11:13:29.309564Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:29.309564Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Embodied Brain: Let Multimodal Large Language Models See, Think, and Control in Spaces","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Erfei Cui, Ganlin Yang, Gen Luo, Guanzhou Chen, Haonan Duan, Jifeng Dai, Jingbo Wang, Lewei Lu, Ronglei Tong, Rongrong Ji, Shenglong Ye, Tianyi Zhang, Wenhai Wang, Xizhou Zhu, Yu Qiao, Zhe Chen, Zhi Hou, Ziyang Gong","submitted_at":"2025-05-30T18:00:34Z","abstract_excerpt":"The remarkable progress of Multimodal Large Language Models (MLLMs) has attracted increasing attention to extend them to physical entities like legged robot. This typically requires MLLMs to not only grasp multimodal understanding abilities, but also integrate visual-spatial reasoning and physical interaction capabilities. Nevertheless,existing methods struggle to unify these capabilities due to their fundamental differences.In this paper, we present the Visual Embodied Brain (VeBrain), a unified framework for perception, reasoning, and control in real world. VeBrain reformulates robotic contr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00123","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00123/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00123","created_at":"2026-07-05T11:13:29.309622+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00123v1","created_at":"2026-07-05T11:13:29.309622+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00123","created_at":"2026-07-05T11:13:29.309622+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOYHG2TWNZLJ","created_at":"2026-07-05T11:13:29.309622+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOYHG2TWNZLJXXDO","created_at":"2026-07-05T11:13:29.309622+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOYHG2TW","created_at":"2026-07-05T11:13:29.309622+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13040","citing_title":"RoboProcessBench: Benchmarking Process-Aware Understanding in Vision-Language Robotic Manipulation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13497","citing_title":"SPARC: Reliable Spatial Annotations from Robot Demonstrations at Scale","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11324","citing_title":"Embodied-R1.5: Evolving Physical Intelligence via Embodied Foundation Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03890","citing_title":"OVO-S-Bench: A Hierarchical Benchmark for Streaming Spatial Intelligence in Multimodal LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16713","citing_title":"GeoWorld-VLM: Geometry from World Models for Vision-Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22536","citing_title":"SpaceDG: Benchmarking Spatial Intelligence under Visual Degradation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15753","citing_title":"RoboPIN: Grounded Embodied Reasoning via Pinned Chain-of-Thought","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25813","citing_title":"Extending Embodied Question Answering from Perception to Decision","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28548","citing_title":"GEM: Generative Supervision Helps Embodied Intelligence","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2508.12043","citing_title":"Talk Less, Fly Lighter: Autonomous Semantic Compression for UAV Swarm Communication via LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22536","citing_title":"SpaceDG: Benchmarking Spatial Intelligence under Visual Degradation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16713","citing_title":"GeoWorld-VLM: Geometry from World Models for Vision-Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16518","citing_title":"MiMo-Embodied: X-Embodied Foundation Model Technical Report","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13778","citing_title":"InternVLA-M1: A Spatially Guided Vision-Language-Action Framework for Generalist Robot Policy","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02870","citing_title":"Token Warping Helps MLLMs Look from Nearby Viewpoints","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08645","citing_title":"3D-VCD: Hallucination Mitigation in 3D-LLM Embodied Agents through Visual Contrastive Decoding","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX","json":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX.json","graph_json":"https://pith.science/api/pith-number/IOYHG2TWNZLJXXDOOZPWSZJHAX/graph.json","events_json":"https://pith.science/api/pith-number/IOYHG2TWNZLJXXDOOZPWSZJHAX/events.json","paper":"https://pith.science/paper/IOYHG2TW"},"agent_actions":{"view_html":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX","download_json":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX.json","view_paper":"https://pith.science/paper/IOYHG2TW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00123&json=true","fetch_graph":"https://pith.science/api/pith-number/IOYHG2TWNZLJXXDOOZPWSZJHAX/graph.json","fetch_events":"https://pith.science/api/pith-number/IOYHG2TWNZLJXXDOOZPWSZJHAX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX/action/storage_attestation","attest_author":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX/action/author_attestation","sign_citation":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX/action/citation_signature","submit_replication":"https://pith.science/pith/IOYHG2TWNZLJXXDOOZPWSZJHAX/action/replication_record"}},"created_at":"2026-07-05T11:13:29.309622+00:00","updated_at":"2026-07-05T11:13:29.309622+00:00"}