{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6IJ53V2PZ7UF56G3UDYAXUCLNJ","short_pith_number":"pith:6IJ53V2P","schema_version":"1.0","canonical_sha256":"f213ddd74fcfe85ef8dba0f00bd04b6a67e6e346cb5e448dff6480679346dd8e","source":{"kind":"arxiv","id":"2307.08581","version":1},"attestation_state":"computed","paper":{"title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Daquan Zhou, Jiashi Feng, Yang Zhao, Zhijie Lin, Zilong Huang","submitted_at":"2023-07-17T15:51:47Z","abstract_excerpt":"LLMs have demonstrated remarkable abilities at interacting with humans through language, especially with the usage of instruction-following data. Recent advancements in LLMs, such as MiniGPT-4, LLaVA, and X-LLM, further enlarge their abilities by incorporating multi-modal inputs, including image, video, and speech. Despite their effectiveness at generating precise and detailed language understanding of the given modality signal, these LLMs give up the ability to ground specific parts of inputs, thus only constructing a coarse-grained mapping. However, explicit and informative correspondence be"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.08581","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-07-17T15:51:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"76f57aa7fc60263e9b12324ba029c466fe501e7b0e35fc2a63fba82a49f638df","abstract_canon_sha256":"2bcde901db28a24e6d39543e952f199d4d0f854172c25733c87ccf94a2dabca0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:31:26.061892Z","signature_b64":"I/UQkKfvpEiljs5T0q3V+Y9mTrA+FI2O4WKULwoZwf3s9TlM5zvBhuCCWyWDgtKtFMjBDKc9QB2INdFCq8d/DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f213ddd74fcfe85ef8dba0f00bd04b6a67e6e346cb5e448dff6480679346dd8e","last_reissued_at":"2026-07-05T06:31:26.061436Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:31:26.061436Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BuboGPT: Enabling Visual Grounding in Multi-Modal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Daquan Zhou, Jiashi Feng, Yang Zhao, Zhijie Lin, Zilong Huang","submitted_at":"2023-07-17T15:51:47Z","abstract_excerpt":"LLMs have demonstrated remarkable abilities at interacting with humans through language, especially with the usage of instruction-following data. Recent advancements in LLMs, such as MiniGPT-4, LLaVA, and X-LLM, further enlarge their abilities by incorporating multi-modal inputs, including image, video, and speech. Despite their effectiveness at generating precise and detailed language understanding of the given modality signal, these LLMs give up the ability to ground specific parts of inputs, thus only constructing a coarse-grained mapping. However, explicit and informative correspondence be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.08581","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.08581/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.08581","created_at":"2026-07-05T06:31:26.061491+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.08581v1","created_at":"2026-07-05T06:31:26.061491+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.08581","created_at":"2026-07-05T06:31:26.061491+00:00"},{"alias_kind":"pith_short_12","alias_value":"6IJ53V2PZ7UF","created_at":"2026-07-05T06:31:26.061491+00:00"},{"alias_kind":"pith_short_16","alias_value":"6IJ53V2PZ7UF56G3","created_at":"2026-07-05T06:31:26.061491+00:00"},{"alias_kind":"pith_short_8","alias_value":"6IJ53V2P","created_at":"2026-07-05T06:31:26.061491+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06534","citing_title":"CAIRN: Cross-Room 3D Scene Understanding with Topology-Aware Large Multimodal Models","ref_index":63,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00592","citing_title":"Through the PRISM: Principle-Aware, Interpretable, and Multi-Scale Evaluation of Visual Designs","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2401.01614","citing_title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27507","citing_title":"Chat-Scene++: Exploiting Context-Rich Object Identification for 3D LLM","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18562","citing_title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","ref_index":175,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ","json":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ.json","graph_json":"https://pith.science/api/pith-number/6IJ53V2PZ7UF56G3UDYAXUCLNJ/graph.json","events_json":"https://pith.science/api/pith-number/6IJ53V2PZ7UF56G3UDYAXUCLNJ/events.json","paper":"https://pith.science/paper/6IJ53V2P"},"agent_actions":{"view_html":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ","download_json":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ.json","view_paper":"https://pith.science/paper/6IJ53V2P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.08581&json=true","fetch_graph":"https://pith.science/api/pith-number/6IJ53V2PZ7UF56G3UDYAXUCLNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/6IJ53V2PZ7UF56G3UDYAXUCLNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ/action/storage_attestation","attest_author":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ/action/author_attestation","sign_citation":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ/action/citation_signature","submit_replication":"https://pith.science/pith/6IJ53V2PZ7UF56G3UDYAXUCLNJ/action/replication_record"}},"created_at":"2026-07-05T06:31:26.061491+00:00","updated_at":"2026-07-05T06:31:26.061491+00:00"}