{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YPZQFC5WVBMPCHGIQJ7H6EZIU7","short_pith_number":"pith:YPZQFC5W","schema_version":"1.0","canonical_sha256":"c3f3028bb6a858f11cc8827e7f1328a7ecc2908872fd6857b055f264c90b677a","source":{"kind":"arxiv","id":"2305.05662","version":4},"attestation_state":"computed","paper":{"title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiashuo Yu, Jifeng Dai, Kunchang Li, Limin Wang, Ping Luo, Qinglong Zhang, Qingyun Li, Shoufa Chen, Weiyun Wang, Wenhai Wang, Xizhou Zhu, Xue Yang, Yali Wang, Yang Yang, Yinan He, Yi Wang, Yu Qiao, Zeqiang Lai, Zhaoyang Liu, Zhe Chen","submitted_at":"2023-05-09T17:58:34Z","abstract_excerpt":"We present an interactive visual framework named InternGPT, or iGPT for short. The framework integrates chatbots that have planning and reasoning capabilities, such as ChatGPT, with non-verbal instructions like pointing movements that enable users to directly manipulate images or videos on the screen. Pointing (including gestures, cursors, etc.) movements can provide more flexibility and precision in performing vision-centric tasks that require fine-grained control, editing, and generation of visual content. The name InternGPT stands for \\textbf{inter}action, \\textbf{n}onverbal, and \\textbf{ch"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.05662","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-05-09T17:58:34Z","cross_cats_sorted":[],"title_canon_sha256":"729a4519ad58a0f1ecf2e61a4715b68d1717413573b015a1fc6d2aef7ee03ffa","abstract_canon_sha256":"b71f0df86dadbf4cb6561f3e9d4fbab1344130fc90ec7c5368f6b261f8c99a05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:50.239238Z","signature_b64":"ORhv4/7vE3BT0quuvcghHr6D3Ma89bce5oEpGOjnPYXuE1uG/FbjWZnROnmHO17uAd1JH8zzIly1HlmSc4A1Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3f3028bb6a858f11cc8827e7f1328a7ecc2908872fd6857b055f264c90b677a","last_reissued_at":"2026-07-05T06:16:50.238802Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:50.238802Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiashuo Yu, Jifeng Dai, Kunchang Li, Limin Wang, Ping Luo, Qinglong Zhang, Qingyun Li, Shoufa Chen, Weiyun Wang, Wenhai Wang, Xizhou Zhu, Xue Yang, Yali Wang, Yang Yang, Yinan He, Yi Wang, Yu Qiao, Zeqiang Lai, Zhaoyang Liu, Zhe Chen","submitted_at":"2023-05-09T17:58:34Z","abstract_excerpt":"We present an interactive visual framework named InternGPT, or iGPT for short. The framework integrates chatbots that have planning and reasoning capabilities, such as ChatGPT, with non-verbal instructions like pointing movements that enable users to directly manipulate images or videos on the screen. Pointing (including gestures, cursors, etc.) movements can provide more flexibility and precision in performing vision-centric tasks that require fine-grained control, editing, and generation of visual content. The name InternGPT stands for \\textbf{inter}action, \\textbf{n}onverbal, and \\textbf{ch"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.05662","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.05662/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.05662","created_at":"2026-07-05T06:16:50.238859+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.05662v4","created_at":"2026-07-05T06:16:50.238859+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.05662","created_at":"2026-07-05T06:16:50.238859+00:00"},{"alias_kind":"pith_short_12","alias_value":"YPZQFC5WVBMP","created_at":"2026-07-05T06:16:50.238859+00:00"},{"alias_kind":"pith_short_16","alias_value":"YPZQFC5WVBMPCHGI","created_at":"2026-07-05T06:16:50.238859+00:00"},{"alias_kind":"pith_short_8","alias_value":"YPZQFC5W","created_at":"2026-07-05T06:16:50.238859+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17539","citing_title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12555","citing_title":"AudioX-Turbo: A Unified Framework for Efficient Anything-to-Audio Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":181,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15947","citing_title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06942","citing_title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2305.06355","citing_title":"VideoChat: Chat-Centric Video Understanding","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03231","citing_title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":299,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15670","citing_title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7","json":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7.json","graph_json":"https://pith.science/api/pith-number/YPZQFC5WVBMPCHGIQJ7H6EZIU7/graph.json","events_json":"https://pith.science/api/pith-number/YPZQFC5WVBMPCHGIQJ7H6EZIU7/events.json","paper":"https://pith.science/paper/YPZQFC5W"},"agent_actions":{"view_html":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7","download_json":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7.json","view_paper":"https://pith.science/paper/YPZQFC5W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.05662&json=true","fetch_graph":"https://pith.science/api/pith-number/YPZQFC5WVBMPCHGIQJ7H6EZIU7/graph.json","fetch_events":"https://pith.science/api/pith-number/YPZQFC5WVBMPCHGIQJ7H6EZIU7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7/action/storage_attestation","attest_author":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7/action/author_attestation","sign_citation":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7/action/citation_signature","submit_replication":"https://pith.science/pith/YPZQFC5WVBMPCHGIQJ7H6EZIU7/action/replication_record"}},"created_at":"2026-07-05T06:16:50.238859+00:00","updated_at":"2026-07-05T06:16:50.238859+00:00"}