{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PJSIZCY6MVAZPDRC3SLQY5QZXY","short_pith_number":"pith:PJSIZCY6","schema_version":"1.0","canonical_sha256":"7a648c8b1e6541978e22dc970c7619be35e32f27d4e3653396874806b8fa2748","source":{"kind":"arxiv","id":"2405.15684","version":1},"attestation_state":"computed","paper":{"title":"Prompt-Aware Adapter: Towards Learning Adaptive Visual Tokens for Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Hehe Fan, Yi Yang, Yue Zhang","submitted_at":"2024-05-24T16:24:10Z","abstract_excerpt":"To bridge the gap between vision and language modalities, Multimodal Large Language Models (MLLMs) usually learn an adapter that converts visual inputs to understandable tokens for Large Language Models (LLMs). However, most adapters generate consistent visual tokens, regardless of the specific objects of interest mentioned in the prompt. Since these adapters distribute equal attention to every detail in the image and focus on the entire scene, they may increase the cognitive load for LLMs, particularly when processing complex scenes. To alleviate this problem, we propose prompt-aware adapters"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15684","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-24T16:24:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ec991e477cfd6aeaaa74a71a82a50317a5b7dfe754bf93848319f790aa379601","abstract_canon_sha256":"f697c7754138f56fe20049cb46ab4f5a5f04ae6bcd227551669441c385fa4901"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:52.601084Z","signature_b64":"dfTdv9uIsVwzQXpBe3H14IbUmo0F+lBwL7ohnLquMZ/pJsbHtEMeL6ZHVYy9u0e08fsFG1w2/1/Fwh9TJ6UoAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a648c8b1e6541978e22dc970c7619be35e32f27d4e3653396874806b8fa2748","last_reissued_at":"2026-07-05T08:22:52.600560Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:52.600560Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prompt-Aware Adapter: Towards Learning Adaptive Visual Tokens for Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Hehe Fan, Yi Yang, Yue Zhang","submitted_at":"2024-05-24T16:24:10Z","abstract_excerpt":"To bridge the gap between vision and language modalities, Multimodal Large Language Models (MLLMs) usually learn an adapter that converts visual inputs to understandable tokens for Large Language Models (LLMs). However, most adapters generate consistent visual tokens, regardless of the specific objects of interest mentioned in the prompt. Since these adapters distribute equal attention to every detail in the image and focus on the entire scene, they may increase the cognitive load for LLMs, particularly when processing complex scenes. To alleviate this problem, we propose prompt-aware adapters"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15684","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15684","created_at":"2026-07-05T08:22:52.600617+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15684v1","created_at":"2026-07-05T08:22:52.600617+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15684","created_at":"2026-07-05T08:22:52.600617+00:00"},{"alias_kind":"pith_short_12","alias_value":"PJSIZCY6MVAZ","created_at":"2026-07-05T08:22:52.600617+00:00"},{"alias_kind":"pith_short_16","alias_value":"PJSIZCY6MVAZPDRC","created_at":"2026-07-05T08:22:52.600617+00:00"},{"alias_kind":"pith_short_8","alias_value":"PJSIZCY6","created_at":"2026-07-05T08:22:52.600617+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.00260","citing_title":"Instruction-Grounded Visual Projectors for Continual Learning of Generative Vision-Language Models","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY","json":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY.json","graph_json":"https://pith.science/api/pith-number/PJSIZCY6MVAZPDRC3SLQY5QZXY/graph.json","events_json":"https://pith.science/api/pith-number/PJSIZCY6MVAZPDRC3SLQY5QZXY/events.json","paper":"https://pith.science/paper/PJSIZCY6"},"agent_actions":{"view_html":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY","download_json":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY.json","view_paper":"https://pith.science/paper/PJSIZCY6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15684&json=true","fetch_graph":"https://pith.science/api/pith-number/PJSIZCY6MVAZPDRC3SLQY5QZXY/graph.json","fetch_events":"https://pith.science/api/pith-number/PJSIZCY6MVAZPDRC3SLQY5QZXY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY/action/storage_attestation","attest_author":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY/action/author_attestation","sign_citation":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY/action/citation_signature","submit_replication":"https://pith.science/pith/PJSIZCY6MVAZPDRC3SLQY5QZXY/action/replication_record"}},"created_at":"2026-07-05T08:22:52.600617+00:00","updated_at":"2026-07-05T08:22:52.600617+00:00"}