{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:V53RYMLOXK6TOLML52FZ4CQNAQ","short_pith_number":"pith:V53RYMLO","schema_version":"1.0","canonical_sha256":"af771c316ebabd372d8bee8b9e0a0d043bc775e9b3aaf26218739d5a3576f59c","source":{"kind":"arxiv","id":"2505.12884","version":2},"attestation_state":"computed","paper":{"title":"TinyAlign: Boosting Lightweight Vision-Language Models by Mitigating Modal Alignment Bottlenecks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Gen Li, Jin Dong, Kejian Wu, Wenjun Wu, Xiaotie Deng, Xinyu Wang, Ye Qiu, Yifan Sun, Yuanze Hu, Zhaoxin Fan, Zhichao Yang","submitted_at":"2025-05-19T09:11:54Z","abstract_excerpt":"Lightweight Vision-Language Models (VLMs) are indispensable for resource-constrained applications. The prevailing approach to aligning vision and language models involves freezing both the vision encoder and the language model while training small connector modules. However, this strategy heavily depends on the intrinsic capabilities of the language model, which can be suboptimal for lightweight models with limited representational capacity. In this work, we investigate this alignment bottleneck through the lens of mutual information, demonstrating that the constrained capacity of the language"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12884","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-19T09:11:54Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"9888fc4ac4887f179459a0bc7894c94180c684e1ac41231d5228bcb59250fa53","abstract_canon_sha256":"609cb5c92ad0fab16b85be0baf0b2f36b92787eefb68acc40c943c1fd6b2808b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:11.570981Z","signature_b64":"jLG+ZLJrYEL/bXki/x1kvNmWqOvt/2To58UJgMzcvz3prSFixhMxRJ5L7/4zdWI6hGsf0iQAvlM3dJeFld1UAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af771c316ebabd372d8bee8b9e0a0d043bc775e9b3aaf26218739d5a3576f59c","last_reissued_at":"2026-07-05T11:29:11.570415Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:11.570415Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TinyAlign: Boosting Lightweight Vision-Language Models by Mitigating Modal Alignment Bottlenecks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Gen Li, Jin Dong, Kejian Wu, Wenjun Wu, Xiaotie Deng, Xinyu Wang, Ye Qiu, Yifan Sun, Yuanze Hu, Zhaoxin Fan, Zhichao Yang","submitted_at":"2025-05-19T09:11:54Z","abstract_excerpt":"Lightweight Vision-Language Models (VLMs) are indispensable for resource-constrained applications. The prevailing approach to aligning vision and language models involves freezing both the vision encoder and the language model while training small connector modules. However, this strategy heavily depends on the intrinsic capabilities of the language model, which can be suboptimal for lightweight models with limited representational capacity. In this work, we investigate this alignment bottleneck through the lens of mutual information, demonstrating that the constrained capacity of the language"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12884","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12884/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12884","created_at":"2026-07-05T11:29:11.570479+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12884v2","created_at":"2026-07-05T11:29:11.570479+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12884","created_at":"2026-07-05T11:29:11.570479+00:00"},{"alias_kind":"pith_short_12","alias_value":"V53RYMLOXK6T","created_at":"2026-07-05T11:29:11.570479+00:00"},{"alias_kind":"pith_short_16","alias_value":"V53RYMLOXK6TOLML","created_at":"2026-07-05T11:29:11.570479+00:00"},{"alias_kind":"pith_short_8","alias_value":"V53RYMLO","created_at":"2026-07-05T11:29:11.570479+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04627","citing_title":"MIRAGE: Mobile Agents with Implicit Reasoning and Generative World Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26614","citing_title":"State Beyond Appearance: Diagnosing and Improving State Consistency in Dial-Based Measurement Reading","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ","json":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ.json","graph_json":"https://pith.science/api/pith-number/V53RYMLOXK6TOLML52FZ4CQNAQ/graph.json","events_json":"https://pith.science/api/pith-number/V53RYMLOXK6TOLML52FZ4CQNAQ/events.json","paper":"https://pith.science/paper/V53RYMLO"},"agent_actions":{"view_html":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ","download_json":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ.json","view_paper":"https://pith.science/paper/V53RYMLO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12884&json=true","fetch_graph":"https://pith.science/api/pith-number/V53RYMLOXK6TOLML52FZ4CQNAQ/graph.json","fetch_events":"https://pith.science/api/pith-number/V53RYMLOXK6TOLML52FZ4CQNAQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ/action/storage_attestation","attest_author":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ/action/author_attestation","sign_citation":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ/action/citation_signature","submit_replication":"https://pith.science/pith/V53RYMLOXK6TOLML52FZ4CQNAQ/action/replication_record"}},"created_at":"2026-07-05T11:29:11.570479+00:00","updated_at":"2026-07-05T11:29:11.570479+00:00"}