{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I3ABJOIWQFDQ4C5X27IXNDZLZT","short_pith_number":"pith:I3ABJOIW","schema_version":"1.0","canonical_sha256":"46c014b91681470e0bb7d7d1768f2bccfcff1e862e2b9482293ab9f3893aed29","source":{"kind":"arxiv","id":"2407.21534","version":6},"attestation_state":"computed","paper":{"title":"ControlMLLM: Training-Free Visual Prompt Learning for Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gen Luo, Guannan Jiang, Hao Fei, Jiale Li, Jiayi Ji, Mingrui Wu, Oucheng Huang, Rongrong Ji, Xiaoshuai Sun, Xinyue Cai","submitted_at":"2024-07-31T11:40:29Z","abstract_excerpt":"In this work, we propose a training-free method to inject visual prompts into Multimodal Large Language Models (MLLMs) through test-time optimization of a learnable latent variable. We observe that attention, as the core module of MLLMs, connects text prompt tokens and visual tokens, ultimately determining the final results. Our approach involves adjusting visual tokens from the MLP output at test time, controlling the attention response to ensure text prompt tokens attend to visual tokens in referring regions. We optimize a learnable latent variable based on an energy function, enhancing the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.21534","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-31T11:40:29Z","cross_cats_sorted":[],"title_canon_sha256":"a5014bf5f86fa38674da6367bd18c2c1e14e527a10d48bf8d296fac7838fa628","abstract_canon_sha256":"ba3d31133129b1177b588098e5a31107dc500410141446d8e6265f9f50a6fa5f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:57:44.508304Z","signature_b64":"OZfxcrJhrYNCgYwcM/vNmn292qrzqnPvpReIa+r7AkwrzRj4dRq2lAdjGubcCKFde03Hzlypf7RxCYNIoWLJAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"46c014b91681470e0bb7d7d1768f2bccfcff1e862e2b9482293ab9f3893aed29","last_reissued_at":"2026-07-05T09:57:44.507852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:57:44.507852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ControlMLLM: Training-Free Visual Prompt Learning for Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gen Luo, Guannan Jiang, Hao Fei, Jiale Li, Jiayi Ji, Mingrui Wu, Oucheng Huang, Rongrong Ji, Xiaoshuai Sun, Xinyue Cai","submitted_at":"2024-07-31T11:40:29Z","abstract_excerpt":"In this work, we propose a training-free method to inject visual prompts into Multimodal Large Language Models (MLLMs) through test-time optimization of a learnable latent variable. We observe that attention, as the core module of MLLMs, connects text prompt tokens and visual tokens, ultimately determining the final results. Our approach involves adjusting visual tokens from the MLP output at test time, controlling the attention response to ensure text prompt tokens attend to visual tokens in referring regions. We optimize a learnable latent variable based on an energy function, enhancing the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.21534","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.21534/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.21534","created_at":"2026-07-05T09:57:44.507909+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.21534v6","created_at":"2026-07-05T09:57:44.507909+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.21534","created_at":"2026-07-05T09:57:44.507909+00:00"},{"alias_kind":"pith_short_12","alias_value":"I3ABJOIWQFDQ","created_at":"2026-07-05T09:57:44.507909+00:00"},{"alias_kind":"pith_short_16","alias_value":"I3ABJOIWQFDQ4C5X","created_at":"2026-07-05T09:57:44.507909+00:00"},{"alias_kind":"pith_short_8","alias_value":"I3ABJOIW","created_at":"2026-07-05T09:57:44.507909+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.04670","citing_title":"Are They the Same? Exploring Visual Correspondence Shortcomings of Multimodal LLMs","ref_index":88,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT","json":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT.json","graph_json":"https://pith.science/api/pith-number/I3ABJOIWQFDQ4C5X27IXNDZLZT/graph.json","events_json":"https://pith.science/api/pith-number/I3ABJOIWQFDQ4C5X27IXNDZLZT/events.json","paper":"https://pith.science/paper/I3ABJOIW"},"agent_actions":{"view_html":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT","download_json":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT.json","view_paper":"https://pith.science/paper/I3ABJOIW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.21534&json=true","fetch_graph":"https://pith.science/api/pith-number/I3ABJOIWQFDQ4C5X27IXNDZLZT/graph.json","fetch_events":"https://pith.science/api/pith-number/I3ABJOIWQFDQ4C5X27IXNDZLZT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT/action/storage_attestation","attest_author":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT/action/author_attestation","sign_citation":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT/action/citation_signature","submit_replication":"https://pith.science/pith/I3ABJOIWQFDQ4C5X27IXNDZLZT/action/replication_record"}},"created_at":"2026-07-05T09:57:44.507909+00:00","updated_at":"2026-07-05T09:57:44.507909+00:00"}