{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ONHBQAABWPFC4TBGYXZ4BOS3FZ","short_pith_number":"pith:ONHBQAAB","schema_version":"1.0","canonical_sha256":"734e180001b3ca2e4c26c5f3c0ba5b2e441fdc35706399dc056a127c6d52b3cd","source":{"kind":"arxiv","id":"2408.12637","version":1},"attestation_state":"computed","paper":{"title":"Building and better understanding vision-language models: insights and future directions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Andr\\'es Marafioti, Hugo Lauren\\c{c}on, L\\'eo Tronchon, Victor Sanh","submitted_at":"2024-08-22T17:47:24Z","abstract_excerpt":"The field of vision-language models (VLMs), which take images and texts as inputs and output texts, is rapidly evolving and has yet to reach consensus on several key aspects of the development pipeline, including data, architecture, and training methods. This paper can be seen as a tutorial for building a VLM. We begin by providing a comprehensive overview of the current state-of-the-art approaches, highlighting the strengths and weaknesses of each, addressing the major challenges in the field, and suggesting promising research directions for underexplored areas. We then walk through the pract"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.12637","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-22T17:47:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e201cedace8d47459abe3112ecf76917b2f929fb08c4ec934374491fd4921e2f","abstract_canon_sha256":"e0dc0a1ee166e404aab9294b7e02ec7549bdcb27824d6adaba3d17eefb767817"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:27.702635Z","signature_b64":"b81TWq2dIivQ6lbUB5/yCGAUhySMrSZuuZFPahdalQtSfYktv+vMM5pmHRHujKrlZ9L9V+GHMSams0J6c3M5BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"734e180001b3ca2e4c26c5f3c0ba5b2e441fdc35706399dc056a127c6d52b3cd","last_reissued_at":"2026-07-05T08:58:27.702117Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:27.702117Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Building and better understanding vision-language models: insights and future directions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Andr\\'es Marafioti, Hugo Lauren\\c{c}on, L\\'eo Tronchon, Victor Sanh","submitted_at":"2024-08-22T17:47:24Z","abstract_excerpt":"The field of vision-language models (VLMs), which take images and texts as inputs and output texts, is rapidly evolving and has yet to reach consensus on several key aspects of the development pipeline, including data, architecture, and training methods. This paper can be seen as a tutorial for building a VLM. We begin by providing a comprehensive overview of the current state-of-the-art approaches, highlighting the strengths and weaknesses of each, addressing the major challenges in the field, and suggesting promising research directions for underexplored areas. We then walk through the pract"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.12637","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.12637/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.12637","created_at":"2026-07-05T08:58:27.702181+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.12637v1","created_at":"2026-07-05T08:58:27.702181+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.12637","created_at":"2026-07-05T08:58:27.702181+00:00"},{"alias_kind":"pith_short_12","alias_value":"ONHBQAABWPFC","created_at":"2026-07-05T08:58:27.702181+00:00"},{"alias_kind":"pith_short_16","alias_value":"ONHBQAABWPFC4TBG","created_at":"2026-07-05T08:58:27.702181+00:00"},{"alias_kind":"pith_short_8","alias_value":"ONHBQAAB","created_at":"2026-07-05T08:58:27.702181+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19960","citing_title":"Stellar: Scalable Multimodal Document Retrieval for Natural Language Queries","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00390","citing_title":"Zamba2-VL Technical Report","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25952","citing_title":"VEN-VL: A Visual Ensemble MoE Framework for Effective and Efficient Multi-Modal Understanding","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04468","citing_title":"NVILA: Efficient Frontier Visual Language Models","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2508.07630","citing_title":"InterChart: Benchmarking Visual Reasoning Across Decomposed and Distributed Chart Information","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22151","citing_title":"MultiMat: Multimodal Program Synthesis for Procedural Materials using Large Multimodal Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07605","citing_title":"Fine-R1: Make Multi-modal LLMs Excel in Fine-Grained Visual Recognition by Chain-of-Thought Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2409.17146","citing_title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11405","citing_title":"20/20 Vision Language Models: A Prescription for Better VLMs through Data Curation Alone","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02813","citing_title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2504.05299","citing_title":"SmolVLM: Redefining small and efficient multimodal models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11405","citing_title":"20/20 Vision Language Models: A Prescription for Better VLMs through Data Curation Alone","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08560","citing_title":"ZAYA1-VL-8B Technical Report","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10985","citing_title":"Back to the Barn with LLAMAs: Evolving Pretrained LLM Backbones in Finetuning Vision Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16099","citing_title":"DenTab: A Dataset for Table Recognition and Visual QA on Real-World Dental Estimates","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17570","citing_title":"PBSBench: A Multi-Level Vision-Language Framework and Benchmark for Hematopathology Whole Slide Image Interpretation","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ","json":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ.json","graph_json":"https://pith.science/api/pith-number/ONHBQAABWPFC4TBGYXZ4BOS3FZ/graph.json","events_json":"https://pith.science/api/pith-number/ONHBQAABWPFC4TBGYXZ4BOS3FZ/events.json","paper":"https://pith.science/paper/ONHBQAAB"},"agent_actions":{"view_html":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ","download_json":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ.json","view_paper":"https://pith.science/paper/ONHBQAAB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.12637&json=true","fetch_graph":"https://pith.science/api/pith-number/ONHBQAABWPFC4TBGYXZ4BOS3FZ/graph.json","fetch_events":"https://pith.science/api/pith-number/ONHBQAABWPFC4TBGYXZ4BOS3FZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ/action/storage_attestation","attest_author":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ/action/author_attestation","sign_citation":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ/action/citation_signature","submit_replication":"https://pith.science/pith/ONHBQAABWPFC4TBGYXZ4BOS3FZ/action/replication_record"}},"created_at":"2026-07-05T08:58:27.702181+00:00","updated_at":"2026-07-05T08:58:27.702181+00:00"}