{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:S6KUP7KDP52VLDFWL7KWP54UXH","short_pith_number":"pith:S6KUP7KD","schema_version":"1.0","canonical_sha256":"979547fd437f75558cb65fd567f794b9de80573990c042e776f51cdd93d6382b","source":{"kind":"arxiv","id":"2307.04087","version":3},"attestation_state":"computed","paper":{"title":"SVIT: Scaling up Visual Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boya Wu, Bo Zhao, Muyang He, Tiejun Huang","submitted_at":"2023-07-09T03:25:14Z","abstract_excerpt":"Thanks to the emerging of foundation models, the large language and vision models are integrated to acquire the multimodal ability of visual captioning, question answering, etc. Although existing multimodal models present impressive performance of visual understanding and reasoning, their limits are still largely under-explored due to the scarcity of high-quality instruction tuning data. To push the limits of multimodal capability, we Scale up Visual Instruction Tuning (SVIT) by constructing a dataset of 4.2 million visual instruction tuning data including 1.6M conversation question-answer (QA"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.04087","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-07-09T03:25:14Z","cross_cats_sorted":[],"title_canon_sha256":"fef5fd14ea0f950ed54f56492cddfdbb3416604d136c8b610d3d500488c5590a","abstract_canon_sha256":"91a5620e97570e98120b25c40766901de764f4b2311a332b6d99f2f816a2ae7f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:28.844580Z","signature_b64":"jspCPR7L70qg+1ret7sqkeIBcS5N96qkSJLrtzrro7YIori9B4UPNksyY+0xpMqAsRo3S+9xYmaLwv7LY4kTBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"979547fd437f75558cb65fd567f794b9de80573990c042e776f51cdd93d6382b","last_reissued_at":"2026-07-05T07:28:28.844057Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:28.844057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SVIT: Scaling up Visual Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boya Wu, Bo Zhao, Muyang He, Tiejun Huang","submitted_at":"2023-07-09T03:25:14Z","abstract_excerpt":"Thanks to the emerging of foundation models, the large language and vision models are integrated to acquire the multimodal ability of visual captioning, question answering, etc. Although existing multimodal models present impressive performance of visual understanding and reasoning, their limits are still largely under-explored due to the scarcity of high-quality instruction tuning data. To push the limits of multimodal capability, we Scale up Visual Instruction Tuning (SVIT) by constructing a dataset of 4.2 million visual instruction tuning data including 1.6M conversation question-answer (QA"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.04087","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.04087/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.04087","created_at":"2026-07-05T07:28:28.844129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.04087v3","created_at":"2026-07-05T07:28:28.844129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.04087","created_at":"2026-07-05T07:28:28.844129+00:00"},{"alias_kind":"pith_short_12","alias_value":"S6KUP7KDP52V","created_at":"2026-07-05T07:28:28.844129+00:00"},{"alias_kind":"pith_short_16","alias_value":"S6KUP7KDP52VLDFW","created_at":"2026-07-05T07:28:28.844129+00:00"},{"alias_kind":"pith_short_8","alias_value":"S6KUP7KD","created_at":"2026-07-05T07:28:28.844129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00465","citing_title":"StochasT: Learning with Stochastic Turn Depth for Visual Instruction Tuning","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05552","citing_title":"Balancing Image Compression and Generation with Bootstrapped Tokenization","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2308.12067","citing_title":"MM-LIMA: Less Is More for Alignment in Multi-Modal Datasets","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04468","citing_title":"NVILA: Efficient Frontier Visual Language Models","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17726","citing_title":"Slot-MLLM: Object-Centric Visual Tokenization for Multimodal LLM","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21300","citing_title":"Reducing Object Hallucination in LVLMs via Emphasizing Image-negative Tokens","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2408.04840","citing_title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03766","citing_title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2311.04257","citing_title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16502","citing_title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2305.03726","citing_title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02813","citing_title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2310.03744","citing_title":"Improved Baselines with Visual Instruction Tuning","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03426","citing_title":"Replacing Parameters with Preferences: Federated Alignment of Heterogeneous Vision-Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01800","citing_title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02589","citing_title":"Representation learning from OCT images","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH","json":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH.json","graph_json":"https://pith.science/api/pith-number/S6KUP7KDP52VLDFWL7KWP54UXH/graph.json","events_json":"https://pith.science/api/pith-number/S6KUP7KDP52VLDFWL7KWP54UXH/events.json","paper":"https://pith.science/paper/S6KUP7KD"},"agent_actions":{"view_html":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH","download_json":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH.json","view_paper":"https://pith.science/paper/S6KUP7KD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.04087&json=true","fetch_graph":"https://pith.science/api/pith-number/S6KUP7KDP52VLDFWL7KWP54UXH/graph.json","fetch_events":"https://pith.science/api/pith-number/S6KUP7KDP52VLDFWL7KWP54UXH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH/action/storage_attestation","attest_author":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH/action/author_attestation","sign_citation":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH/action/citation_signature","submit_replication":"https://pith.science/pith/S6KUP7KDP52VLDFWL7KWP54UXH/action/replication_record"}},"created_at":"2026-07-05T07:28:28.844129+00:00","updated_at":"2026-07-05T07:28:28.844129+00:00"}