{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K63M7MBTZRFKBUXE737MY3O6PA","short_pith_number":"pith:K63M7MBT","schema_version":"1.0","canonical_sha256":"57b6cfb033cc4aa0d2e4fefecc6dde780b83801988abe6cc7982b3b4ce9cc08f","source":{"kind":"arxiv","id":"2308.04152","version":4},"attestation_state":"computed","paper":{"title":"Fine-tuning Multimodal LLMs to Follow Zero-shot Demonstrative Instructions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Juncheng Li, Kaihang Pan, Minghe Gao, Siliang Tang, Tat-Seng Chua, Wei Ji, Wenqiao Zhang, Yueting Zhuang, Zhiqi Ge","submitted_at":"2023-08-08T09:32:43Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) have been utilizing Visual Prompt Generators (VPGs) to convert visual features into tokens that LLMs can recognize. This is achieved by training the VPGs on millions of image-caption pairs, where the VPG-generated tokens of images are fed into a frozen LLM to generate the corresponding captions. However, this image-captioning based training objective inherently biases the VPG to concentrate solely on the primary visual contents sufficient for caption generation, often neglecting other visual details. This shortcoming results in ML"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.04152","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-08T09:32:43Z","cross_cats_sorted":[],"title_canon_sha256":"e815bf9710c1429ff411e9eadbcc3db410b167dfe8514d84ed1260146ca8e88a","abstract_canon_sha256":"e6059f092995f6484d8f1ce09ed1809fb0e52c7d96e4128420079f6bb8ba2752"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:11.197247Z","signature_b64":"8q+BDF16Z95dGsCvn0clLTpUBgbOjgaXJrljxpr9snGPDo7a5QLLh+mMbMb/1iaUFa6H9RgWx6krTtLKoQFGDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57b6cfb033cc4aa0d2e4fefecc6dde780b83801988abe6cc7982b3b4ce9cc08f","last_reissued_at":"2026-07-05T08:23:11.196753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:11.196753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-tuning Multimodal LLMs to Follow Zero-shot Demonstrative Instructions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Juncheng Li, Kaihang Pan, Minghe Gao, Siliang Tang, Tat-Seng Chua, Wei Ji, Wenqiao Zhang, Yueting Zhuang, Zhiqi Ge","submitted_at":"2023-08-08T09:32:43Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) have been utilizing Visual Prompt Generators (VPGs) to convert visual features into tokens that LLMs can recognize. This is achieved by training the VPGs on millions of image-caption pairs, where the VPG-generated tokens of images are fed into a frozen LLM to generate the corresponding captions. However, this image-captioning based training objective inherently biases the VPG to concentrate solely on the primary visual contents sufficient for caption generation, often neglecting other visual details. This shortcoming results in ML"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.04152","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.04152/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.04152","created_at":"2026-07-05T08:23:11.196813+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.04152v4","created_at":"2026-07-05T08:23:11.196813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.04152","created_at":"2026-07-05T08:23:11.196813+00:00"},{"alias_kind":"pith_short_12","alias_value":"K63M7MBTZRFK","created_at":"2026-07-05T08:23:11.196813+00:00"},{"alias_kind":"pith_short_16","alias_value":"K63M7MBTZRFKBUXE","created_at":"2026-07-05T08:23:11.196813+00:00"},{"alias_kind":"pith_short_8","alias_value":"K63M7MBT","created_at":"2026-07-05T08:23:11.196813+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2309.15112","citing_title":"InternLM-XComposer: A Vision-Language Large Model for Advanced Text-image Comprehension and Composition","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13394","citing_title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2408.03326","citing_title":"LLaVA-OneVision: Easy Visual Task Transfer","ref_index":70,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA","json":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA.json","graph_json":"https://pith.science/api/pith-number/K63M7MBTZRFKBUXE737MY3O6PA/graph.json","events_json":"https://pith.science/api/pith-number/K63M7MBTZRFKBUXE737MY3O6PA/events.json","paper":"https://pith.science/paper/K63M7MBT"},"agent_actions":{"view_html":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA","download_json":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA.json","view_paper":"https://pith.science/paper/K63M7MBT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.04152&json=true","fetch_graph":"https://pith.science/api/pith-number/K63M7MBTZRFKBUXE737MY3O6PA/graph.json","fetch_events":"https://pith.science/api/pith-number/K63M7MBTZRFKBUXE737MY3O6PA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA/action/storage_attestation","attest_author":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA/action/author_attestation","sign_citation":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA/action/citation_signature","submit_replication":"https://pith.science/pith/K63M7MBTZRFKBUXE737MY3O6PA/action/replication_record"}},"created_at":"2026-07-05T08:23:11.196813+00:00","updated_at":"2026-07-05T08:23:11.196813+00:00"}