{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2SWNAFTDUUKOD4VF7QORDFWFE3","short_pith_number":"pith:2SWNAFTD","schema_version":"1.0","canonical_sha256":"d4acd01663a514e1f2a5fc1d1196c526c9640b98e469a8b6927ae60c2a5524fe","source":{"kind":"arxiv","id":"2307.09474","version":1},"attestation_state":"computed","paper":{"title":"ChatSpot: Bootstrapping Multimodal LLMs via Precise Referring Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Chunrui Han, En Yu, Haoran Wei, Hongyu Zhou, Jianjian Sun, Jinrong Yang, Liang Zhao, Runpei Dong, Xiangyu Zhang, Yuang Peng, Zheng Ge","submitted_at":"2023-07-18T17:56:06Z","abstract_excerpt":"Human-AI interactivity is a critical aspect that reflects the usability of multimodal large language models (MLLMs). However, existing end-to-end MLLMs only allow users to interact with them through language instructions, leading to the limitation of the interactive accuracy and efficiency. In this study, we present precise referring instructions that utilize diverse reference representations such as points and boxes as referring prompts to refer to the special region. This enables MLLMs to focus on the region of interest and achieve finer-grained interaction. Based on precise referring instru"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.09474","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-18T17:56:06Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"b2fe06473f6d2d20c67c0a942df623e1ce64d052daf1f903900b05345111ffe1","abstract_canon_sha256":"3c74415689e37dabae14171e3b96ae0f4ba5ead2df08dec86dda458517f6ba83"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:32:17.266490Z","signature_b64":"URCbor65GwEGE4RiCCRMJsHFKaeqqoxP4haCTZBUQs7obcvVtQHqJfkuNGTla59Woa7aISwdVZkdWjR08AIiDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4acd01663a514e1f2a5fc1d1196c526c9640b98e469a8b6927ae60c2a5524fe","last_reissued_at":"2026-07-05T06:32:17.266025Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:32:17.266025Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChatSpot: Bootstrapping Multimodal LLMs via Precise Referring Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Chunrui Han, En Yu, Haoran Wei, Hongyu Zhou, Jianjian Sun, Jinrong Yang, Liang Zhao, Runpei Dong, Xiangyu Zhang, Yuang Peng, Zheng Ge","submitted_at":"2023-07-18T17:56:06Z","abstract_excerpt":"Human-AI interactivity is a critical aspect that reflects the usability of multimodal large language models (MLLMs). However, existing end-to-end MLLMs only allow users to interact with them through language instructions, leading to the limitation of the interactive accuracy and efficiency. In this study, we present precise referring instructions that utilize diverse reference representations such as points and boxes as referring prompts to refer to the special region. This enables MLLMs to focus on the region of interest and achieve finer-grained interaction. Based on precise referring instru"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.09474","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.09474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.09474","created_at":"2026-07-05T06:32:17.266085+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.09474v1","created_at":"2026-07-05T06:32:17.266085+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.09474","created_at":"2026-07-05T06:32:17.266085+00:00"},{"alias_kind":"pith_short_12","alias_value":"2SWNAFTDUUKO","created_at":"2026-07-05T06:32:17.266085+00:00"},{"alias_kind":"pith_short_16","alias_value":"2SWNAFTDUUKOD4VF","created_at":"2026-07-05T06:32:17.266085+00:00"},{"alias_kind":"pith_short_8","alias_value":"2SWNAFTD","created_at":"2026-07-05T06:32:17.266085+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09132","citing_title":"Vision Language Model Helps Private Information De-Identification in Vision Data","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18018","citing_title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01402","citing_title":"Injecting Distributional Awareness into MLLMs via Reinforcement Learning for Deep Imbalanced Regression","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01402","citing_title":"Injecting Distributional Awareness into MLLMs via Reinforcement Learning for Deep Imbalanced Regression","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11789","citing_title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","ref_index":236,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3","json":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3.json","graph_json":"https://pith.science/api/pith-number/2SWNAFTDUUKOD4VF7QORDFWFE3/graph.json","events_json":"https://pith.science/api/pith-number/2SWNAFTDUUKOD4VF7QORDFWFE3/events.json","paper":"https://pith.science/paper/2SWNAFTD"},"agent_actions":{"view_html":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3","download_json":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3.json","view_paper":"https://pith.science/paper/2SWNAFTD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.09474&json=true","fetch_graph":"https://pith.science/api/pith-number/2SWNAFTDUUKOD4VF7QORDFWFE3/graph.json","fetch_events":"https://pith.science/api/pith-number/2SWNAFTDUUKOD4VF7QORDFWFE3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3/action/storage_attestation","attest_author":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3/action/author_attestation","sign_citation":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3/action/citation_signature","submit_replication":"https://pith.science/pith/2SWNAFTDUUKOD4VF7QORDFWFE3/action/replication_record"}},"created_at":"2026-07-05T06:32:17.266085+00:00","updated_at":"2026-07-05T06:32:17.266085+00:00"}