{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PDY5DBBSZ77XSPNESALVWYJCWS","short_pith_number":"pith:PDY5DBBS","schema_version":"1.0","canonical_sha256":"78f1d18432cfff793da490175b6122b48f10e85cc6c62ec3aecf8ed6d668fbf9","source":{"kind":"arxiv","id":"2305.01278","version":2},"attestation_state":"computed","paper":{"title":"VPGTrans: Transfer Visual Prompt Generator across LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Ao Zhang, Hao Fei, Li Li, Tat-Seng Chua, Wei Ji, Yuan Yao, Zhiyuan Liu","submitted_at":"2023-05-02T09:28:39Z","abstract_excerpt":"While developing a new multimodal LLM (MLLM) by pre-training on tremendous image-text pairs from scratch can be exceedingly resource-consuming, connecting an existing LLM with a comparatively lightweight visual prompt generator (VPG) becomes a feasible paradigm. However, further tuning the VPG part of the MLLM still suffers from indispensable computational costs, i.e., requiring thousands of GPU hours and millions of training data. One alternative solution is to transfer an existing VPG from any existing MLLMs for the target MLLM.\n  In this work, we for the first time investigate the VPG trans"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.01278","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-02T09:28:39Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"6e1a83bd2834a58290940f2b3852f2a2c3f40f2e37938010232103d4daf57c46","abstract_canon_sha256":"84dcdb5970ed82c5ccb5b6028f876b1c4907733f3e2896d63aaab025b131ccb6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:04:15.022138Z","signature_b64":"vrF4Soo1SQ5RmP+HRmgOxAUZK8gt+Q1UVjXpTi5/Rrh26AImzU/UDtKhc7XRyh57ewmjOwCceg4zvfJBkv2PAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78f1d18432cfff793da490175b6122b48f10e85cc6c62ec3aecf8ed6d668fbf9","last_reissued_at":"2026-07-05T07:04:15.021698Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:04:15.021698Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VPGTrans: Transfer Visual Prompt Generator across LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Ao Zhang, Hao Fei, Li Li, Tat-Seng Chua, Wei Ji, Yuan Yao, Zhiyuan Liu","submitted_at":"2023-05-02T09:28:39Z","abstract_excerpt":"While developing a new multimodal LLM (MLLM) by pre-training on tremendous image-text pairs from scratch can be exceedingly resource-consuming, connecting an existing LLM with a comparatively lightweight visual prompt generator (VPG) becomes a feasible paradigm. However, further tuning the VPG part of the MLLM still suffers from indispensable computational costs, i.e., requiring thousands of GPU hours and millions of training data. One alternative solution is to transfer an existing VPG from any existing MLLMs for the target MLLM.\n  In this work, we for the first time investigate the VPG trans"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.01278","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.01278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.01278","created_at":"2026-07-05T07:04:15.021754+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.01278v2","created_at":"2026-07-05T07:04:15.021754+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.01278","created_at":"2026-07-05T07:04:15.021754+00:00"},{"alias_kind":"pith_short_12","alias_value":"PDY5DBBSZ77X","created_at":"2026-07-05T07:04:15.021754+00:00"},{"alias_kind":"pith_short_16","alias_value":"PDY5DBBSZ77XSPNE","created_at":"2026-07-05T07:04:15.021754+00:00"},{"alias_kind":"pith_short_8","alias_value":"PDY5DBBS","created_at":"2026-07-05T07:04:15.021754+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2303.16199","citing_title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13394","citing_title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS","json":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS.json","graph_json":"https://pith.science/api/pith-number/PDY5DBBSZ77XSPNESALVWYJCWS/graph.json","events_json":"https://pith.science/api/pith-number/PDY5DBBSZ77XSPNESALVWYJCWS/events.json","paper":"https://pith.science/paper/PDY5DBBS"},"agent_actions":{"view_html":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS","download_json":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS.json","view_paper":"https://pith.science/paper/PDY5DBBS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.01278&json=true","fetch_graph":"https://pith.science/api/pith-number/PDY5DBBSZ77XSPNESALVWYJCWS/graph.json","fetch_events":"https://pith.science/api/pith-number/PDY5DBBSZ77XSPNESALVWYJCWS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS/action/storage_attestation","attest_author":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS/action/author_attestation","sign_citation":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS/action/citation_signature","submit_replication":"https://pith.science/pith/PDY5DBBSZ77XSPNESALVWYJCWS/action/replication_record"}},"created_at":"2026-07-05T07:04:15.021754+00:00","updated_at":"2026-07-05T07:04:15.021754+00:00"}