{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:33MWATZYLXTWI6HEUXN2A3TYHJ","short_pith_number":"pith:33MWATZY","schema_version":"1.0","canonical_sha256":"ded9604f385de76478e4a5dba06e783a706e0e57dace8d5e1ed9941aebd5e6b3","source":{"kind":"arxiv","id":"2409.02958","version":1},"attestation_state":"computed","paper":{"title":"Multi-Modal Adapter for Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dominykas Seputis, Serghei Mihailov, Soham Chatterjee, Zehao Xiao","submitted_at":"2024-09-03T12:47:08Z","abstract_excerpt":"Large pre-trained vision-language models, such as CLIP, have demonstrated state-of-the-art performance across a wide range of image classification tasks, without requiring retraining. Few-shot CLIP is competitive with existing specialized architectures that were trained on the downstream tasks. Recent research demonstrates that the performance of CLIP can be further improved using lightweight adaptation approaches. However, previous methods adapt different modalities of the CLIP model individually, ignoring the interactions and relationships between visual and textual representations. In this "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.02958","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-03T12:47:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a9a90af77a1982851b30a39f12e5dd9792a754249269fa431ad6505da062b21d","abstract_canon_sha256":"f62621e559960776b0199f88708fa14f6a604eb17996a029be26bf96e9a22e6c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:03:24.409031Z","signature_b64":"OA2nZgfP76hL1iNJ/iprexSFhu+HV2E634K6n+OnctKT5KEJzDKj+7IPFmYki+LWNox8m3OZKOSReUg7WukuAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ded9604f385de76478e4a5dba06e783a706e0e57dace8d5e1ed9941aebd5e6b3","last_reissued_at":"2026-07-05T09:03:24.408516Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:03:24.408516Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Modal Adapter for Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dominykas Seputis, Serghei Mihailov, Soham Chatterjee, Zehao Xiao","submitted_at":"2024-09-03T12:47:08Z","abstract_excerpt":"Large pre-trained vision-language models, such as CLIP, have demonstrated state-of-the-art performance across a wide range of image classification tasks, without requiring retraining. Few-shot CLIP is competitive with existing specialized architectures that were trained on the downstream tasks. Recent research demonstrates that the performance of CLIP can be further improved using lightweight adaptation approaches. However, previous methods adapt different modalities of the CLIP model individually, ignoring the interactions and relationships between visual and textual representations. In this "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.02958","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.02958/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.02958","created_at":"2026-07-05T09:03:24.408580+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.02958v1","created_at":"2026-07-05T09:03:24.408580+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.02958","created_at":"2026-07-05T09:03:24.408580+00:00"},{"alias_kind":"pith_short_12","alias_value":"33MWATZYLXTW","created_at":"2026-07-05T09:03:24.408580+00:00"},{"alias_kind":"pith_short_16","alias_value":"33MWATZYLXTWI6HE","created_at":"2026-07-05T09:03:24.408580+00:00"},{"alias_kind":"pith_short_8","alias_value":"33MWATZY","created_at":"2026-07-05T09:03:24.408580+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.03980","citing_title":"Gram-Anchored Prompt Learning for Vision-Language Models via Second-Order Statistics","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ","json":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ.json","graph_json":"https://pith.science/api/pith-number/33MWATZYLXTWI6HEUXN2A3TYHJ/graph.json","events_json":"https://pith.science/api/pith-number/33MWATZYLXTWI6HEUXN2A3TYHJ/events.json","paper":"https://pith.science/paper/33MWATZY"},"agent_actions":{"view_html":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ","download_json":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ.json","view_paper":"https://pith.science/paper/33MWATZY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.02958&json=true","fetch_graph":"https://pith.science/api/pith-number/33MWATZYLXTWI6HEUXN2A3TYHJ/graph.json","fetch_events":"https://pith.science/api/pith-number/33MWATZYLXTWI6HEUXN2A3TYHJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ/action/storage_attestation","attest_author":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ/action/author_attestation","sign_citation":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ/action/citation_signature","submit_replication":"https://pith.science/pith/33MWATZYLXTWI6HEUXN2A3TYHJ/action/replication_record"}},"created_at":"2026-07-05T09:03:24.408580+00:00","updated_at":"2026-07-05T09:03:24.408580+00:00"}