{"as_of":"2026-08-19T03:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:81e6a80630c06ddc2f5eaf5ebadafd65b3d8d2504747933d371a66edda4e1928","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":48,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:49:10.837634Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T05:39:39.660194Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2409.02813","last_updated":"2025-05-22T08:22:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-04T15:31:26Z","title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-14T00:51:48.163349Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2409.02813"},"observation_digest":"sha256:125becaad8bc817ba50316b436fa7efc3c1ee28cfa578b042755fc1b77b87f54","observation_id":"0f304a0f-a3ca-4e71-9e9a-db73504ec9c7","resolution":{"observed_at":"2026-05-14T00:51:48.342889Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-15T01:55:12.501409Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2409.17146"},"observation_digest":"sha256:3cefeb64c06a14c46044d9e7481a8511aebdb13f97956b3e30a78c3cb05a7344","observation_id":"33e2b272-aef7-43ed-a3d4-b2fd70c979bb","resolution":{"observed_at":"2026-05-15T01:55:12.630563Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-12T15:17:06.804474Z","title":"Building and better understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-08-17T16:49:55.394145Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:17:06.804474Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2411.14432"},"observation_digest":"sha256:076b67b4fe0935f2a71e7be58650f7bc5d32199b2876a614a36b347234aae114","observation_id":"92ab786f-14a3-4d3f-91b0-43fa501d2784","resolution":{"observed_at":"2026-08-12T15:17:06.804474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-12T10:36:35.917708Z","title":"Building and bet- ter understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.19103","last_updated":"2024-11-28T12:38:42Z","snapshot_observed_at":"2026-08-18T01:37:50.499305Z","submitted_at":"2024-11-28T12:38:42Z","title":"VARCO-VISION: Expanding Frontiers in Korean Vision-Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T10:36:35.917708Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2411.19103"},"observation_digest":"sha256:6de6e4413650b84caafe37534f436313265970096fba841a709b882014281d6d","observation_id":"e481b500-b1fb-47f6-a568-685b8dc5b950","resolution":{"observed_at":"2026-08-12T10:36:35.917708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2412.04468","last_updated":"2026-04-25T07:16:42Z","snapshot_observed_at":"2026-08-03T08:48:57.969106Z","submitted_at":"2024-12-05T18:59:55Z","title":"NVILA: Efficient Frontier Visual Language Models","version":3},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-23T07:42:22.478647Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2412.04468"},"observation_digest":"sha256:f373a88d25112b437174a982db060c0fdf3343ddc4c8388325f1b7645355881b","observation_id":"2ca4f020-57c9-4508-95f9-499a99c5fe63","resolution":{"observed_at":"2026-05-23T07:42:43.129247Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":121,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:4bae2025f059e74291e87787c4a285c76c4cf4b6f346bb6da6c246f98de7dd51","observation_id":"7fe45a35-7a91-445e-a1e1-5e349c1c25dd","resolution":{"observed_at":"2026-05-10T13:23:58.029371Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-11T16:11:10.490240Z","title":"Building and better understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-15T17:40:10.309874Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T16:11:10.490240Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2412.10360"},"observation_digest":"sha256:b99e2e3e2f88b67abc125ec426ed1081237e7bf384d3ff768290f867a62f1e3c","observation_id":"020c5bb4-6836-4b3f-9270-e1dfc5e9856d","resolution":{"observed_at":"2026-08-11T16:11:10.490240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-11T10:43:08.131404Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16364","last_updated":"2024-12-20T21:55:15Z","snapshot_observed_at":"2026-08-15T05:20:03.750113Z","submitted_at":"2024-12-20T21:55:15Z","title":"A High-Quality Text-Rich Image Instruction Tuning Dataset via Hybrid Instruction Generation","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T10:43:08.131404Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2412.16364"},"observation_digest":"sha256:99ed3d5ba5c0cc1d6f0b7e00d15bd3a613b275ed69e6485000b1deea8dcaab2f","observation_id":"4bad5c29-2f67-490c-be1d-f7bb08f05aa9","resolution":{"observed_at":"2026-08-11T10:43:08.131404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-10T22:41:33.498202Z","title":"Building and better understanding vision- language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.00958","last_updated":"2025-05-13T16:29:08Z","snapshot_observed_at":"2026-08-15T17:14:28.671540Z","submitted_at":"2025-01-01T21:29:37Z","title":"2.5 Years in Class: A Multimodal Textbook for Vision-Language Pretraining","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T22:41:33.498202Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2501.00958"},"observation_digest":"sha256:4b01b2347edd8006bebeb4419abdb49d8f026204c0bd1978d44dbd3f7593fa87","observation_id":"4662635b-f70f-440c-bc44-1957f06b8306","resolution":{"observed_at":"2026-08-10T22:41:33.498202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-10T21:57:14.213819Z","title":"Building and better understanding vision- language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.03225","last_updated":"2025-04-09T17:25:07Z","snapshot_observed_at":"2026-08-14T01:18:40.798180Z","submitted_at":"2025-01-06T18:57:31Z","title":"Automated Generation of Challenging Multiple-Choice Questions for Vision Language Model Evaluation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T21:57:14.213819Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2501.03225"},"observation_digest":"sha256:59a4677a43c394a7edd05b6710079c6d1037e6058302628a16af2825d5731af8","observation_id":"627db8c1-9b90-473b-b6c5-3bff1107c01a","resolution":{"observed_at":"2026-08-10T21:57:14.213819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-10T21:25:47.061351Z","title":"Building and better understanding vision- language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05031","last_updated":"2025-03-13T07:45:55Z","snapshot_observed_at":"2026-08-16T08:14:12.523999Z","submitted_at":"2025-01-09T07:43:49Z","title":"ECBench: Can Multi-modal Foundation Models Understand the Egocentric World? A Holistic Embodied Cognition Benchmark","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T21:25:47.061351Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2501.05031"},"observation_digest":"sha256:d307abd13bc3bd221840479178773d0a288da74e2b7ee17fe98196ac2a3a23e9","observation_id":"c7969d33-d702-405b-9d60-6ab0036c8e96","resolution":{"observed_at":"2026-08-10T21:25:47.061351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-10T19:27:42.242753Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.10057","last_updated":"2025-01-17T09:22:35Z","snapshot_observed_at":"2026-08-17T16:43:21.444345Z","submitted_at":"2025-01-17T09:22:35Z","title":"MSTS: A Multimodal Safety Test Suite for Vision-Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T19:27:42.242753Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2501.10057"},"observation_digest":"sha256:1d15944d7832efc37b58bce561e437751e45e1c476906c2d005424590a77128a","observation_id":"5e4f1832-6a6e-4fa8-8245-cedc8457168b","resolution":{"observed_at":"2026-08-10T19:27:42.242753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-10T18:04:34.402880Z","title":"Building and better understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.14818","last_updated":"2025-01-20T18:40:47Z","snapshot_observed_at":"2026-08-11T01:59:38.596539Z","submitted_at":"2025-01-20T18:40:47Z","title":"Eagle 2: Building Post-Training Data Strategies from Scratch for Frontier Vision-Language Models","version":1},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:34.402880Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2501.14818"},"observation_digest":"sha256:79a9dd5ef7433984e7d4376c656052ceb6d3a3a66da6647b7c1069ebc2ea0bb5","observation_id":"39a1499e-823e-4baa-96a7-1963287bdc7f","resolution":{"observed_at":"2026-08-10T18:04:34.402880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-09T19:15:41.568116Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00385","last_updated":"2025-02-26T03:23:41Z","snapshot_observed_at":"2026-08-12T11:36:29.082052Z","submitted_at":"2025-02-01T09:53:17Z","title":"The Impact of Persona-based Political Perspectives on Hateful Content Detection","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-09T19:15:41.568116Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2502.00385"},"observation_digest":"sha256:a1f002daf6d119571bf234bda7ba3888dd1689959024cb5aadb92cd8523b5414","observation_id":"603ad2fc-c9ac-432a-963e-be588269380b","resolution":{"observed_at":"2026-08-09T19:15:41.568116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-08T14:25:55.706584Z","title":"Building and better understanding vision- language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06788","last_updated":"2025-07-24T10:29:52Z","snapshot_observed_at":"2026-08-09T02:10:24.105469Z","submitted_at":"2025-02-10T18:59:58Z","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-08T14:25:55.706584Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2502.06788"},"observation_digest":"sha256:7015776045c3dee4643c6880f23a60e3697ebc35b3e77ba4ee83e476fa57d10c","observation_id":"f78ef67f-3e12-4473-8176-d60d129373ad","resolution":{"observed_at":"2026-08-08T14:25:55.706584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2504.05299","last_updated":"2025-04-07T17:58:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T17:58:57Z","title":"SmolVLM: Redefining small and efficient multimodal models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T20:23:50.552549Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2504.05299"},"observation_digest":"sha256:06c6df7aebd4c6bbf752a776434e5d3bb0b5f1a86ab0e97cf1c44fb8215b1db6","observation_id":"10b5a689-0ad0-401e-a7ac-e47ca724df58","resolution":{"observed_at":"2026-05-13T20:23:51.725976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-16T10:49:10.837634Z","title":"Building and better understanding vision-language models: insights and future directions.CoRR, abs/2408.12637,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18589","last_updated":"2025-05-13T06:32:02Z","snapshot_observed_at":"2026-08-18T02:43:22.898514Z","submitted_at":"2025-04-24T06:16:38Z","title":"Benchmarking Multimodal Mathematical Reasoning with Explicit Visual Dependency","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T10:49:10.837634Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2504.18589"},"observation_digest":"sha256:ef79841c8aed8c7dd16617fbe19761baa617b06dd26d8a12e557cf2bbd3afb12","observation_id":"cb1f77ae-97fa-40b4-93a4-fe4f374436a2","resolution":{"observed_at":"2026-08-16T10:49:10.837634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-15T23:41:34.161930Z","title":"Building and better understanding vision- language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.04147","last_updated":"2025-05-07T05:55:45Z","snapshot_observed_at":"2026-08-17T13:04:30.039874Z","submitted_at":"2025-05-07T05:55:45Z","title":"R^3-VQA: \"Read the Room\" by Video Social Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T23:41:34.161930Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2505.04147"},"observation_digest":"sha256:5702dd8fbff3a6f257c38a5e5de5609b55ad35bff38897527d28473f63894745","observation_id":"cfe53a72-f40e-43b5-b681-e17e28e4a155","resolution":{"observed_at":"2026-08-15T23:41:34.161930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-15T23:17:58.708306Z","title":"Building and better understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.05071","last_updated":"2025-05-21T06:18:41Z","snapshot_observed_at":"2026-08-18T22:45:00.629363Z","submitted_at":"2025-05-08T09:06:53Z","title":"FG-CLIP: Fine-Grained Visual and Textual Alignment","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T23:17:58.708306Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2505.05071"},"observation_digest":"sha256:2b6749dfae3710bc8d7da20a2a47b671ea86bc269b9c2f395b6b9683b5844c45","observation_id":"4dd34dc0-0697-4789-a243-673797ff84f3","resolution":{"observed_at":"2026-08-15T23:17:58.708306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-06T21:52:23.673554Z","title":"Building and better understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23115","last_updated":"2025-06-29T06:41:00Z","snapshot_observed_at":"2026-08-12T19:13:17.249586Z","submitted_at":"2025-06-29T06:41:00Z","title":"MoCa: Modality-aware Continual Pre-training Makes Better Bidirectional Multimodal Embeddings","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.673554Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2506.23115"},"observation_digest":"sha256:eee9b5b464418c8655c7789e706797f0a48585b2ffda00b3a3a5cbb2c38505ca","observation_id":"2021fc53-9986-495d-b081-c67fdc160c3b","resolution":{"observed_at":"2026-08-06T21:52:23.673554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-06T21:52:23.743202Z","title":"URL https://doi.org/10.48550/arXiv.2408","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23115","last_updated":"2025-06-29T06:41:00Z","snapshot_observed_at":"2026-08-12T19:13:17.249586Z","submitted_at":"2025-06-29T06:41:00Z","title":"MoCa: Modality-aware Continual Pre-training Makes Better Bidirectional Multimodal Embeddings","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.743202Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2506.23115"},"observation_digest":"sha256:0532e3768aa549992a2c09036bfd7d0719a31c760b0a046a9b70b4a4861c339b","observation_id":"de154433-0b74-42e1-91f1-83e5f4c6ba35","resolution":{"observed_at":"2026-08-06T21:52:23.743202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-06T21:26:12.795081Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00152","last_updated":"2025-06-30T18:04:36Z","snapshot_observed_at":"2026-08-18T18:59:44.156213Z","submitted_at":"2025-06-30T18:04:36Z","title":"Table Understanding and (Multimodal) LLMs: A Cross-Domain Case Study on Scientific vs. Non-Scientific Data","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T21:26:12.795081Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2507.00152"},"observation_digest":"sha256:7b20502f60e6e91880ab6fef17abe2d91d8ef838eb48e0774de4d6ffebe744b4","observation_id":"8251c337-49a0-4d1b-ba3b-3da9da9008f3","resolution":{"observed_at":"2026-08-06T21:26:12.795081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-06T19:44:39.601722Z","title":"Building and better understanding vision-language models: insights and future direc- tions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04741","last_updated":"2025-07-07T08:16:38Z","snapshot_observed_at":"2026-08-09T18:08:52.487764Z","submitted_at":"2025-07-07T08:16:38Z","title":"Vision-Language Models Can't See the Obvious","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:44:39.601722Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2507.04741"},"observation_digest":"sha256:90ac6e907a1d5bec5f7eeb42560e3fdc24a4d904e9c1a45a6a3a6310a2059eae","observation_id":"280158b9-84d2-4448-93ab-a08974e26520","resolution":{"observed_at":"2026-08-06T19:44:39.601722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-06T15:57:01.585192Z","title":"Building and better understanding vision- language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-17T21:12:15.287539Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:01.585192Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:c66b5e2cd0bc330a383524bf7b1178ae2fc394c182360e2f289522601e29af4b","observation_id":"ed95fb85-610e-4847-a8c5-56b3496cd2ac","resolution":{"observed_at":"2026-08-06T15:57:01.585192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-15T18:20:41.196489Z","title":"Building and better understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18300","last_updated":"2025-07-24T11:05:24Z","snapshot_observed_at":"2026-08-17T22:35:22.675252Z","submitted_at":"2025-07-24T11:05:24Z","title":"LMM-Det: Make Large Multimodal Models Excel in Object Detection","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:41.196489Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2507.18300"},"observation_digest":"sha256:c1612323725ecb9d70935577395af56ca5a18e52f564fe77ba07de35f846108e","observation_id":"d39e0ca0-c515-4ee9-844a-d43e7d6201cd","resolution":{"observed_at":"2026-08-15T18:20:41.196489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-05T22:10:08.145820Z","title":"Building and better understanding vision-language models: insights and future directions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.07493","last_updated":"2025-08-25T01:29:22Z","snapshot_observed_at":"2026-08-19T03:33:04.948769Z","submitted_at":"2025-08-10T21:44:43Z","title":"VisR-Bench: An Empirical Study on Visual Retrieval-Augmented Generation for Multilingual Long Document Understanding","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T22:10:08.145820Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2508.07493"},"observation_digest":"sha256:5cbd267b93b7005a97a6dc056e560ac64b2eee063063e7b5371887e5e7816f5e","observation_id":"e254639f-2e33-484c-87c7-28b6ff9e545a","resolution":{"observed_at":"2026-08-05T22:10:08.145820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2508.07630","last_updated":"2026-05-01T04:34:46Z","snapshot_observed_at":"2026-08-11T08:30:15.243248Z","submitted_at":"2025-08-11T05:19:23Z","title":"InterChart: Benchmarking Visual Reasoning Across Decomposed and Distributed Chart Information","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T00:23:52.475744Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2508.07630"},"observation_digest":"sha256:4bee99c6c915a72cef7e569fc8b360f9180651f7357db3472079b301f190a510","observation_id":"c3a56c70-296c-42e0-bd61-7193d84befbb","resolution":{"observed_at":"2026-05-19T00:26:56.175974Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-04T18:49:39.205111Z","title":"arXiv preprint arXiv:2408.12637 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09658","last_updated":"2026-05-25T08:56:37Z","snapshot_observed_at":"2026-08-19T01:08:52.430878Z","submitted_at":"2025-09-11T17:54:00Z","title":"Measuring Epistemic Humility in Multimodal Large Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T18:49:39.205111Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2509.09658"},"observation_digest":"sha256:54d6f6654034d8203e2c8715bbaf41dfa3f28ef30c4a5d4c1b16b840def33714","observation_id":"887a9bfd-dcdf-4a24-99f5-8af7eb53e7af","resolution":{"observed_at":"2026-08-04T18:49:39.205111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2509.22151","last_updated":"2026-05-13T19:15:11Z","snapshot_observed_at":"2026-08-17T14:24:48.802744Z","submitted_at":"2025-09-26T10:10:25Z","title":"MultiMat: Multimodal Program Synthesis for Procedural Materials using Large Multimodal Models","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T13:41:37.594782Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2509.22151"},"observation_digest":"sha256:5c1a6ea96215ee8d6a4df057ddf59ae35d3805c046e94619b0fefcddc2ad1ef4","observation_id":"73d926d0-48cf-4725-bec6-a78b63643ab4","resolution":{"observed_at":"2026-05-18T13:42:38.772215Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-03T20:41:50.086157Z","title":"Building and better understanding vision-language models: insights and future directions.CoRR, abs/2408.12637, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.19032","last_updated":"2026-07-09T04:07:40Z","snapshot_observed_at":"2026-08-16T06:03:41.986353Z","submitted_at":"2025-11-24T12:07:56Z","title":"Diagnosing Corruption-Induced Reliability Failures in Vision-Language Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T20:41:50.086157Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2511.19032"},"observation_digest":"sha256:34692c5756db54e1f3e9ae4a425cdb9d98b58567b4d4acb72470f0a936cfed83","observation_id":"d11ee3f7-00d1-46f5-8818-e8275eecb668","resolution":{"observed_at":"2026-08-03T20:41:50.086157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-03T04:20:54.248497Z","title":"Building and better understanding vision-language models: insights and future directions.arXiv preprint arXiv:2408.12637, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.05275","last_updated":"2026-07-01T09:38:37Z","snapshot_observed_at":"2026-08-15T20:10:49.368970Z","submitted_at":"2026-02-05T04:01:01Z","title":"Magic-MM-Embedding: Towards Visual-Token-Efficient Universal Multimodal Embedding with MLLMs","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-03T04:20:54.248497Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2602.05275"},"observation_digest":"sha256:2ad80399cb850abf44b600613df7b4624810475446116fa10774a6e434503e92","observation_id":"d1a1b7cc-4327-4d06-8e40-7aaa6efa2abc","resolution":{"observed_at":"2026-08-03T04:20:54.248497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2602.07605","last_updated":"2026-04-27T13:49:08Z","snapshot_observed_at":"2026-08-15T05:01:38.609992Z","submitted_at":"2026-02-07T16:16:51Z","title":"Fine-R1: Make Multi-modal LLMs Excel in Fine-Grained Visual Recognition by Chain-of-Thought Reasoning","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T06:13:33.315525Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2602.07605"},"observation_digest":"sha256:a5710cb08a8cd0cbf34c7ed1751d2b423d979ba03ed5f3bc69176c297088eb8d","observation_id":"8eae8d05-7347-43e8-ae0c-00d41ee2405e","resolution":{"observed_at":"2026-05-16T06:17:26.618597Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2604.10985","last_updated":"2026-04-13T04:44:42Z","snapshot_observed_at":"2026-08-17T18:48:58.929735Z","submitted_at":"2026-04-13T04:44:42Z","title":"Back to the Barn with LLAMAs: Evolving Pretrained LLM Backbones in Finetuning Vision Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T16:07:13.367863Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2604.10985"},"observation_digest":"sha256:163db119d6359abba08bed9c5a196f5ac7c06a0c27fae8d554b9b0847f5a3045","observation_id":"cb705c73-6cf4-46f0-ab9c-b9f4c94fd856","resolution":{"observed_at":"2026-05-11T09:20:59.281336Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2604.16099","last_updated":"2026-04-17T14:33:51Z","snapshot_observed_at":"2026-07-06T23:03:30.601598Z","submitted_at":"2026-04-17T14:33:51Z","title":"DenTab: A Dataset for Table Recognition and Visual QA on Real-World Dental Estimates","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T08:33:00.988234Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2604.16099"},"observation_digest":"sha256:42da9afb0c387d99a3e963cd8840df5dd9943316b3e218ed733b2d9c544a9a74","observation_id":"8e96df26-dce8-45a1-946b-530100747880","resolution":{"observed_at":"2026-05-10T08:53:04.999998Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2604.17570","last_updated":"2026-04-19T18:24:11Z","snapshot_observed_at":"2026-07-06T23:04:37.370465Z","submitted_at":"2026-04-19T18:24:11Z","title":"PBSBench: A Multi-Level Vision-Language Framework and Benchmark for Hematopathology Whole Slide Image Interpretation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T05:53:34.007500Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2604.17570"},"observation_digest":"sha256:decf722df3037a42e3117c38b15be1e7cf8a4e86679c05b945b616f189e00c69","observation_id":"1100dfe0-713e-42a0-aa91-4df2231482c6","resolution":{"observed_at":"2026-05-10T05:56:11.157107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2605.08560","last_updated":"2026-05-08T23:41:13Z","snapshot_observed_at":"2026-08-15T12:35:11.807336Z","submitted_at":"2026-05-08T23:41:13Z","title":"ZAYA1-VL-8B Technical Report","version":1},"reference_index":132,"source":"pdf_text","source_observed_at":"2026-05-12T01:15:16.607346Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2605.08560"},"observation_digest":"sha256:63b9cdc07133f5fcbdaf208e83852bf96ea587f54ae92efa33d198a0b0bbb142","observation_id":"5357719c-62c0-442c-9aa0-19b21d897cfc","resolution":{"observed_at":"2026-05-12T08:21:23.510227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2605.11405","last_updated":"2026-05-13T01:55:26Z","snapshot_observed_at":"2026-08-17T07:51:05.413011Z","submitted_at":"2026-05-12T01:51:03Z","title":"20/20 Vision Language Models: A Prescription for Better VLMs through Data Curation Alone","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T02:52:43.674969Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2605.11405"},"observation_digest":"sha256:22ada1a7654d58b478b747d4c7433b390eb1c3d7633342b90777a049a75299e1","observation_id":"dec68707-2fe8-4e97-ac41-1169b7d858c8","resolution":{"observed_at":"2026-05-13T02:57:09.560253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2605.11405","last_updated":"2026-05-13T01:55:26Z","snapshot_observed_at":"2026-08-17T07:51:05.413011Z","submitted_at":"2026-05-12T01:51:03Z","title":"20/20 Vision Language Models: A Prescription for Better VLMs through Data Curation Alone","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-14T21:28:37.680681Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2605.11405"},"observation_digest":"sha256:577ba0c13539034bc2d9113c53b90dfa1011864dd2557f172676865139132127","observation_id":"878d8f83-82c1-45cb-a155-b8526dfe88ec","resolution":{"observed_at":"2026-05-14T21:29:28.614421Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2605.25952","last_updated":"2026-05-25T15:28:48Z","snapshot_observed_at":"2026-08-14T19:49:49.360616Z","submitted_at":"2026-05-25T15:28:48Z","title":"VEN-VL: A Visual Ensemble MoE Framework for Effective and Efficient Multi-Modal Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T22:24:20.787671Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2605.25952"},"observation_digest":"sha256:35d1023710b84ed751398264dd79d2a404a7dc8a1b3c7fb6fd1c420af035bf47","observation_id":"5b374f96-111f-4955-a00a-e20b6dff8c51","resolution":{"observed_at":"2026-06-29T22:34:02.594785Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2606.00390","last_updated":"2026-05-29T22:12:40Z","snapshot_observed_at":"2026-07-06T23:41:05.637982Z","submitted_at":"2026-05-29T22:12:40Z","title":"Zamba2-VL Technical Report","version":1},"reference_index":114,"source":"pdf_text","source_observed_at":"2026-06-28T22:34:20.970856Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2606.00390"},"observation_digest":"sha256:6cc863f733ec8010556098069c822ad312a5f29731372e48c7bdcef695fd28d2","observation_id":"6c79169a-4d02-4fbe-8b40-9d751204ce42","resolution":{"observed_at":"2026-07-01T19:26:00.658133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2606.19960","last_updated":"2026-06-18T08:57:26Z","snapshot_observed_at":"2026-08-04T03:22:17.443424Z","submitted_at":"2026-06-18T08:57:26Z","title":"Stellar: Scalable Multimodal Document Retrieval for Natural Language Queries","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-26T15:49:49.067128Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2606.19960"},"observation_digest":"sha256:bf68b0cea8cba4a1c88408510e1d69f01a3aa16f88fa70bda12ee92b46581674","observation_id":"8398975b-134e-490a-8000-37d2a02181e8","resolution":{"observed_at":"2026-07-04T05:39:39.661854Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2606.28551","last_updated":"2026-08-10T01:17:41Z","snapshot_observed_at":"2026-08-13T23:28:37.542175Z","submitted_at":"2026-06-26T19:11:29Z","title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","version":1},"reference_index":143,"source":"pdf_text","source_observed_at":"2026-06-30T01:16:16.834861Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2606.28551"},"observation_digest":"sha256:66dc70ce38163239e289f356639ebfe3266288f81ff0a6efbe0dcaac90be5401","observation_id":"13e6751d-735c-4c94-8cd0-5ab2e8566b25","resolution":{"observed_at":"2026-07-01T15:45:47.490405Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":"2408.12637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-04T05:39:39.660194Z","title":"Building and Better Understand- ing Vision-Language Models: Insights and Future Directions","venue":null,"work_id":"54aead52-8316-437d-a6cd-78609af679f2","year":2024},"citing_paper":{"arxiv_id":"2606.28551","last_updated":"2026-08-10T01:17:41Z","snapshot_observed_at":"2026-08-13T23:28:37.542175Z","submitted_at":"2026-06-26T19:11:29Z","title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","version":2},"reference_index":143,"source":"pdf_text","source_observed_at":"2026-07-02T21:10:10.548489Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2606.28551"},"observation_digest":"sha256:2b4a2b378771389d599b3d81c34f2dd68541e1380b4d50b6c63a4f7355363c00","observation_id":"38ab0453-c911-4761-86b7-c1cdacef7631","resolution":{"observed_at":"2026-07-02T21:17:23.854434Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-01T08:37:42.020482Z","title":"Building and better understanding vision-language models: Insights and future directions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21061","last_updated":"2026-07-23T08:52:11Z","snapshot_observed_at":"2026-08-08T01:13:21.444754Z","submitted_at":"2026-07-23T08:52:11Z","title":"MVEI & EmObserver: Empowering MLLM-Oriented Visual Emotional Intelligence via Emotion Statement Judgement","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-01T08:37:42.020482Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2607.21061"},"observation_digest":"sha256:19c0016dd5d2ff0ead349f6587247acfdbfc5bf81441da2565c09ebd67fd8a15","observation_id":"c636bb41-f0a3-415f-b6c6-15f95f16eeb9","resolution":{"observed_at":"2026-08-01T08:37:42.020482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-07-30T13:50:41.061598Z","title":"Building and better understanding vision-language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23870","last_updated":"2026-07-26T22:23:43Z","snapshot_observed_at":"2026-08-18T19:39:15.922564Z","submitted_at":"2026-07-26T22:23:43Z","title":"MulRobBench: A Decision-Level Benchmark for Safe and Security-Policy-Compliant Multimodal UAV Agents","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-30T13:50:41.061598Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2607.23870"},"observation_digest":"sha256:4623d21894645fdce8fd3ab9c5d6702cb438b1624d7df01fc357ec028f1b2838","observation_id":"0cfbcc96-ccc5-4fce-b4bb-fc008816a546","resolution":{"observed_at":"2026-07-30T13:50:41.061598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-12T00:43:45.298580Z","title":"arXiv preprint arXiv:2408.12637 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.07943","last_updated":"2026-08-08T06:07:55Z","snapshot_observed_at":"2026-08-15T00:42:59.445824Z","submitted_at":"2026-08-08T06:07:55Z","title":"Locating Failure in Multi-Page Visually Rich Document Understanding: An Empirical Attribution","version":1},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-08-12T00:43:45.298580Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2608.07943"},"observation_digest":"sha256:c02c25f27e84264956baab4e7b48defb2f3c886c872edf432173b8e2b7eb931c","observation_id":"e363be95-6a85-4242-aa43-7a3606fa7fdd","resolution":{"observed_at":"2026-08-12T00:43:45.298580Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-12T20:26:25.681332Z","title":"arXiv preprint arXiv:2408.12637 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10636","last_updated":"2026-08-11T08:23:39Z","snapshot_observed_at":"2026-08-18T00:46:29.820664Z","submitted_at":"2026-08-11T08:23:39Z","title":"DistilVDR: A Compact End-to-End Visual Document Retriever via Dual-Student Distillation","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-12T20:26:25.681332Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2608.10636"},"observation_digest":"sha256:19aaf0cdbb2e87cb6747fe4759d0af89297811a96c0159b0e78047ce21af08be","observation_id":"22d9cf6a-c6dc-4053-90f6-c999d79303cd","resolution":{"observed_at":"2026-08-12T20:26:25.681332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12637","snapshot_observed_at":"2026-08-14T10:59:51.884916Z","title":"arXiv preprint arXiv:2408.12637 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13441","last_updated":"2026-08-13T16:27:18Z","snapshot_observed_at":"2026-08-16T23:12:57.638070Z","submitted_at":"2026-08-13T16:27:18Z","title":"Edit2TikZ: A Comprehensive and Challenging Benchmark for Scientific Figure Editing with TikZ","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-14T10:59:51.884916Z"},"links":{"cited_paper":"/paper/2408.12637","citing_paper":"/paper/2608.13441"},"observation_digest":"sha256:ac6e6ad816dc41b39b59740d7ce2ec7a62828f1ff234b694b0fdbd1318b27070","observation_id":"aa545273-8786-4b5c-bc0e-179698f21fa7","resolution":{"observed_at":"2026-08-14T10:59:51.884916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.12637/citation-record","integrity":"/paper/2408.12637/integrity","json":"/paper/2408.12637/citation-record.json","paper":"/paper/2408.12637"},"outbound":[],"paper":{"arxiv_id":"2408.12637","last_updated":"2024-08-22T17:47:24Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-18T01:32:03.254930Z","submitted_at":"2024-08-22T17:47:24Z","title":"Building and better understanding vision-language models: insights and future directions"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 48 inbound Pith citation observations for arXiv:2408.12637."}