{"as_of":"2026-08-21T15:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0940f07f7033ffe8996f7bfa683a8aecc1d4a538afa6006359f248b318e82850","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":56,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:10:13.173431Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T15:09:55.297801Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-12T11:19:33.797984Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.18363","last_updated":"2025-03-11T14:19:42Z","snapshot_observed_at":"2026-08-19T21:16:45.185594Z","submitted_at":"2024-11-27T14:11:10Z","title":"ChatRex: Taming Multimodal LLM for Joint Perception and Understanding","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T11:19:33.797984Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2411.18363"},"observation_digest":"sha256:758a13a12e146593afb7258524cb6afc0a18b10c72e7c23f6b24e38b6ebebfc1","observation_id":"df506477-f157-4d6d-9192-7b5a093d8d76","resolution":{"observed_at":"2026-08-12T11:19:33.797984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-12T06:04:21.986845Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.19628","last_updated":"2025-07-25T10:41:09Z","snapshot_observed_at":"2026-08-15T09:35:52.423373Z","submitted_at":"2024-11-29T11:24:23Z","title":"Accelerating Multimodal Large Language Models via Dynamic Visual-Token Exit and the Empirical Findings","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T06:04:21.986845Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2411.19628"},"observation_digest":"sha256:637c79ad32df3e55bad7431c1a42cc64c330638ecf75cf23858c27da4ac7a238","observation_id":"520bca3a-9828-4f0a-8aa0-be5507544d76","resolution":{"observed_at":"2026-08-12T06:04:21.986845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-12T04:35:52.390853Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01289","last_updated":"2024-12-04T09:51:16Z","snapshot_observed_at":"2026-08-16T21:24:36.721317Z","submitted_at":"2024-12-02T09:02:28Z","title":"Enhancing Perception Capabilities of Multimodal LLMs with Training-Free Fusion","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T04:35:52.390853Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.01289"},"observation_digest":"sha256:c27ea53864edc05f16ff99eced5020438949524b6b9b091a9f8c4e97afccb975","observation_id":"a6cc9117-80e7-49c2-b2af-c2da7fac13b2","resolution":{"observed_at":"2026-08-12T04:35:52.390853Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T23:54:23.847163Z","title":"Eagle: Exploring the design space for multi- modal llms with mixture of encoders,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02104","last_updated":"2024-12-03T02:54:31Z","snapshot_observed_at":"2026-08-16T12:56:12.864592Z","submitted_at":"2024-12-03T02:54:31Z","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","version":1},"reference_index":144,"source":"pdf_text","source_observed_at":"2026-08-11T23:54:23.847163Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.02104"},"observation_digest":"sha256:e19813558f475005084c6254a0425bdc03b7b283f719aa963862fb05eec37a22","observation_id":"7aa4b2bf-e988-49dc-b003-ebd1a709932b","resolution":{"observed_at":"2026-08-11T23:54:23.847163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T22:18:01.551747Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.03704","last_updated":"2025-06-30T20:29:34Z","snapshot_observed_at":"2026-08-14T19:57:03.409333Z","submitted_at":"2024-12-04T20:35:07Z","title":"Scaling Inference-Time Search with Vision Value Model for Improved Visual Comprehension","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T22:18:01.551747Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.03704"},"observation_digest":"sha256:43824d994f5d64b8bb024f9f7ad48d3aafde7abb6396606e3ba93915502c6f19","observation_id":"16fb8192-cc40-498d-9fdf-55ca80f0de8c","resolution":{"observed_at":"2026-08-11T22:18:01.551747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T21:37:08.369438Z","title":"Eagle: Ex- ploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.04317","last_updated":"2024-12-05T16:34:07Z","snapshot_observed_at":"2026-08-18T02:44:40.547611Z","submitted_at":"2024-12-05T16:34:07Z","title":"FlashSloth: Lightning Multimodal Large Language Models via Embedded Visual Compression","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T21:37:08.369438Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.04317"},"observation_digest":"sha256:116e20bd5406d99682397ce0cb3ec2d91afdaaef21681f7a8cd3ac88fcc9c9ac","observation_id":"16054fb2-0a0c-4f1b-8662-43a7cd5815a4","resolution":{"observed_at":"2026-08-11T21:37:08.369438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T21:27:52.592158Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.04424","last_updated":"2024-12-05T18:50:39Z","snapshot_observed_at":"2026-08-19T09:30:26.841589Z","submitted_at":"2024-12-05T18:50:39Z","title":"Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T21:27:52.592158Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.04424"},"observation_digest":"sha256:b1717d0181ec6fe183833571862fb56c18ffa61ec159bab1d3df19ce592869e9","observation_id":"09010bec-878a-4b04-b1d3-177e846291a6","resolution":{"observed_at":"2026-08-11T21:27:52.592158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2412.04468","last_updated":"2026-04-25T07:16:42Z","snapshot_observed_at":"2026-08-03T08:48:57.969106Z","submitted_at":"2024-12-05T18:59:55Z","title":"NVILA: Efficient Frontier Visual Language Models","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-23T07:42:22.478647Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.04468"},"observation_digest":"sha256:ff8b8f257163dcf99ad8a0077e70e1178d06eb0341b75d561acda7a7b59fd8e9","observation_id":"16dc2d83-7125-423b-8fce-4acc5191ded3","resolution":{"observed_at":"2026-05-23T07:42:43.210588Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":211,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:5bb1b89438b3543982df50dc48132b5cf6979d799c4aaf171b972f8ed3e6bbff","observation_id":"dd7ba175-5922-443a-84ad-b4cd2b439172","resolution":{"observed_at":"2026-05-10T13:23:57.975453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T17:17:49.402538Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.09168","last_updated":"2024-12-12T10:55:57Z","snapshot_observed_at":"2026-08-19T16:13:49.463844Z","submitted_at":"2024-12-12T10:55:57Z","title":"YingSound: Video-Guided Sound Effects Generation with Multi-modal Chain-of-Thought Controls","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-11T17:17:49.402538Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.09168"},"observation_digest":"sha256:be261d9f1cab559021bd91d696ed2a6a69a2a77d6266be72462ca55bf5445cdf","observation_id":"7fe8678d-227b-4473-9d60-eff082295c9d","resolution":{"observed_at":"2026-08-11T17:17:49.402538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T16:57:06.434821Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.09612","last_updated":"2025-04-01T21:08:07Z","snapshot_observed_at":"2026-08-18T01:48:53.175172Z","submitted_at":"2024-12-12T18:59:40Z","title":"Olympus: A Universal Task Router for Computer Vision Tasks","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-11T16:57:06.434821Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.09612"},"observation_digest":"sha256:a0592fa41293790c5a5d1f72fa4d9d7d329d9158ea3bb18b5c073b5e937f40c9","observation_id":"702ae479-9a6d-4899-9919-2aa79d520fd7","resolution":{"observed_at":"2026-08-11T16:57:06.434821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T16:11:10.608343Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-15T17:40:10.309874Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T16:11:10.608343Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.10360"},"observation_digest":"sha256:d95ef45ddb084ebfa5982e554e6d635e1840c2693b8cf17b03100c7f71298015","observation_id":"3f58044c-919b-41c3-abff-d74cc9aff3dc","resolution":{"observed_at":"2026-08-11T16:11:10.608343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T13:19:23.350189Z","title":"Eagle: Ex- ploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13303","last_updated":"2025-05-15T22:00:19Z","snapshot_observed_at":"2026-08-15T01:32:05.414283Z","submitted_at":"2024-12-17T20:09:55Z","title":"FastVLM: Efficient Vision Encoding for Vision Language Models","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:23.350189Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.13303"},"observation_digest":"sha256:35c40cebb85b8aaa943c9d0f52bf253159c9caa8583b524cc00c4916cbc87889","observation_id":"3b98487b-ab3a-4d1d-a282-817d9a3ba1f5","resolution":{"observed_at":"2026-08-11T13:19:23.350189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T12:47:17.711940Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13845","last_updated":"2025-02-24T03:37:59Z","snapshot_observed_at":"2026-08-17T16:57:34.317634Z","submitted_at":"2024-12-18T13:38:06Z","title":"Do Language Models Understand Time?","version":3},"reference_index":133,"source":"pdf_text","source_observed_at":"2026-08-11T12:47:17.711940Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.13845"},"observation_digest":"sha256:c0952b14676137d45f93af4bae39945347b144bd0c26f45ffa714a6b0898e78f","observation_id":"0bcd8123-4256-460d-8e79-bd6bbf39b835","resolution":{"observed_at":"2026-08-11T12:47:17.711940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T12:46:59.893904Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13871","last_updated":"2025-03-19T10:04:22Z","snapshot_observed_at":"2026-08-11T14:38:15.849933Z","submitted_at":"2024-12-18T14:07:46Z","title":"LLaVA-UHD v2: an MLLM Integrating High-Resolution Semantic Pyramid via Hierarchical Window Transformer","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-11T12:46:59.893904Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.13871"},"observation_digest":"sha256:dc2142fcdc8f08cd4c7bdb365c584a9e5cf1572cbd8a907e613607a07d8feb2e","observation_id":"6ab493a7-7d1a-4b78-ac95-00766ae2826f","resolution":{"observed_at":"2026-08-11T12:46:59.893904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-11T10:43:08.259469Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16364","last_updated":"2024-12-20T21:55:15Z","snapshot_observed_at":"2026-08-15T05:20:03.750113Z","submitted_at":"2024-12-20T21:55:15Z","title":"A High-Quality Text-Rich Image Instruction Tuning Dataset via Hybrid Instruction Generation","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T10:43:08.259469Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2412.16364"},"observation_digest":"sha256:ab2c64b789357ff7f036513df4eb9687ddcf36473b1ca9fd47896befe4205b06","observation_id":"fa9e2839-d2f8-4f32-b1b4-61908a1ba1e7","resolution":{"observed_at":"2026-08-11T10:43:08.259469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-12T17:21:52.298102Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":151,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:77238d9a23c2dfa0b357a0ec61e323399b3d4fb647667b853141a65959f1c2f8","observation_id":"44e47afa-7329-4462-b8c0-5b62c58b3edc","resolution":{"observed_at":"2026-05-17T20:33:26.927854Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-10T22:27:13.606238Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01709","last_updated":"2025-03-18T07:34:44Z","snapshot_observed_at":"2026-08-15T13:36:27.630577Z","submitted_at":"2025-01-03T09:10:34Z","title":"MoVE-KD: Knowledge Distillation for VLMs with Mixture of Visual Encoders","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:13.606238Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2501.01709"},"observation_digest":"sha256:24a2dbb9eae4d51162e01d85e103e544255e89c4b3eada572597b20da485b822","observation_id":"cea5852e-cdf8-4cd4-90d7-0adf3091d588","resolution":{"observed_at":"2026-08-10T22:27:13.606238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-10T21:31:16.232974Z","title":"Eagle: Explor- ing the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04670","last_updated":"2025-07-09T08:04:21Z","snapshot_observed_at":"2026-08-14T03:18:34.834807Z","submitted_at":"2025-01-08T18:30:53Z","title":"Are They the Same? Exploring Visual Correspondence Shortcomings of Multimodal LLMs","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T21:31:16.232974Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2501.04670"},"observation_digest":"sha256:44715d890ba24273453d9b5aed5590a6725306e8f7d65b6291198401ec9a6430","observation_id":"f8093b0b-6d46-4454-a930-67eba09d1e5f","resolution":{"observed_at":"2026-08-10T21:31:16.232974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-10T21:10:19.156640Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05901","last_updated":"2025-01-13T02:34:19Z","snapshot_observed_at":"2026-08-14T04:25:59.263298Z","submitted_at":"2025-01-10T11:53:46Z","title":"Valley2: Exploring Multimodal Models with Scalable Vision-Language Design","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T21:10:19.156640Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2501.05901"},"observation_digest":"sha256:5163b1d5de10b6a667ac01f74f0c0203bb42c541180036b98fbbbaa4cc80a5a3","observation_id":"8c1421ce-8e09-4402-ad0f-3d0f91432404","resolution":{"observed_at":"2026-08-10T21:10:19.156640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-10T18:04:33.906379Z","title":"Model Response Figure 12| Eagle2-9B has strong OCR recognition capabilities","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.14818","last_updated":"2025-01-20T18:40:47Z","snapshot_observed_at":"2026-08-11T01:59:38.596539Z","submitted_at":"2025-01-20T18:40:47Z","title":"Eagle 2: Building Post-Training Data Strategies from Scratch for Frontier Vision-Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:33.906379Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2501.14818"},"observation_digest":"sha256:75d5fd6000de7ec2262f5f915b2b24fae5e77f3970c30af01fa28857047bac3b","observation_id":"443394bb-4e23-447a-b82f-55b45b339c66","resolution":{"observed_at":"2026-08-10T18:04:33.906379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-09T21:55:23.295553Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.18954","last_updated":"2025-01-31T08:27:31Z","snapshot_observed_at":"2026-08-18T18:27:01.689176Z","submitted_at":"2025-01-31T08:27:31Z","title":"LLMDet: Learning Strong Open-Vocabulary Object Detectors under the Supervision of Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-09T21:55:23.295553Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2501.18954"},"observation_digest":"sha256:195db5703b138eddd695dde04c7936c3e3173bb72d58aa74ae1b6e6193699504","observation_id":"685e67aa-bf45-4915-bfc3-5aa41164105c","resolution":{"observed_at":"2026-08-09T21:55:23.295553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-09T14:58:53.820823Z","title":"Eagle: Exploring the design space for multi- modal llms with mixture of encoders","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.01576","last_updated":"2026-06-03T04:58:04Z","snapshot_observed_at":"2026-08-16T16:48:00.858764Z","submitted_at":"2025-02-03T17:59:45Z","title":"Robust-LLaVA: On the Effectiveness of Large-Scale Robust Image Encoders for Multi-modal Large Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-09T14:58:53.820823Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2502.01576"},"observation_digest":"sha256:fafd9e67beb6bce26d81683b31fbe2812325ae5404e99471d8e02d0fd67b7e64","observation_id":"f0251a32-8844-48e6-8b63-e3bcce42e86f","resolution":{"observed_at":"2026-08-09T14:58:53.820823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-09T04:22:42.200920Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.03628","last_updated":"2025-07-01T16:02:21Z","snapshot_observed_at":"2026-08-18T18:57:10.590937Z","submitted_at":"2025-02-05T21:34:02Z","title":"The Hidden Life of Tokens: Reducing Hallucination of Large Vision-Language Models via Visual Information Steering","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-09T04:22:42.200920Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2502.03628"},"observation_digest":"sha256:c1de7627ae4a9217c46d4ba38d401f164e571d160da36124573bba5bbc750bc7","observation_id":"b173ffc3-2fd3-4803-a360-30bbb851d192","resolution":{"observed_at":"2026-08-09T04:22:42.200920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-08T20:04:02.333188Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05178","last_updated":"2025-02-07T18:59:57Z","snapshot_observed_at":"2026-08-16T14:57:26.851711Z","submitted_at":"2025-02-07T18:59:57Z","title":"QLIP: Text-Aligned Visual Tokenization Unifies Auto-Regressive Multimodal Understanding and Generation","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-08T20:04:02.333188Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2502.05178"},"observation_digest":"sha256:828fd8d2b21fdf6ae27caffb5831d167e0b34d146bbe0a66f5cb11f99ed0b419","observation_id":"763974f5-2b29-4718-9563-692d710ec268","resolution":{"observed_at":"2026-08-08T20:04:02.333188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-09T11:20:02.912656Z","title":"Eagle: Exploring the design space for multi- modal llms with mixture of encoders","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06814","last_updated":"2025-05-25T15:41:21Z","snapshot_observed_at":"2026-08-20T06:25:10.529526Z","submitted_at":"2025-02-04T22:20:20Z","title":"Diffusion Instruction Tuning","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-09T11:20:02.912656Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2502.06814"},"observation_digest":"sha256:f60abb1e507a909525e62e8dcf3641be68f5a5c2ef6fa8f2dbf9faaa17f1f2a1","observation_id":"533938b4-b2bb-4b3a-bd32-4cf798a355d5","resolution":{"observed_at":"2026-08-09T11:20:02.912656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T22:51:02.140407Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09051","last_updated":"2025-02-13T08:05:44Z","snapshot_observed_at":"2026-08-16T11:45:49.640236Z","submitted_at":"2025-02-13T08:05:44Z","title":"AIDE: Agentically Improve Visual Language Model with Domain Experts","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T22:51:02.140407Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2502.09051"},"observation_digest":"sha256:e67826cfefb8a63ce7c46f1b0d7fbf568a6e17f91dea0ac07ba4a8a55012f8be","observation_id":"9e50171f-9fac-4c3c-a3d1-e69360510242","resolution":{"observed_at":"2026-08-07T22:51:02.140407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2504.09925","last_updated":"2026-04-29T06:12:36Z","snapshot_observed_at":"2026-08-16T10:58:08.667292Z","submitted_at":"2025-04-14T06:33:29Z","title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-22T19:49:00.961388Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2504.09925"},"observation_digest":"sha256:c5d2a46ef4947519c31c86ba65aa2e85f7b4b06560df4cde0ab34dd8e8a839be","observation_id":"9fdb54f3-82dd-45cc-baa4-6bc6b85d3de0","resolution":{"observed_at":"2026-05-22T19:52:01.857966Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-17T09:56:52.502317Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:4e2ee5c8fec6f2b9a0b2ca3a2545345d1e5fb8d3668fb718b0bd22a7553a9e3d","observation_id":"d77b2a76-1a39-403b-ba68-6524d9aab019","resolution":{"observed_at":"2026-05-10T13:41:08.080193Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2504.21850","last_updated":"2026-05-14T18:32:02Z","snapshot_observed_at":"2026-08-16T08:46:08.684546Z","submitted_at":"2025-04-30T17:57:22Z","title":"Visual Compositional Tuning","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-22T17:39:09.890605Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2504.21850"},"observation_digest":"sha256:e7921a4de1d1a5ffb6081b94718f5076bd127278e0576b579def27ff45e04196","observation_id":"1f2af3d1-7069-4150-abb6-d75a83746ec6","resolution":{"observed_at":"2026-05-22T17:41:53.159814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T14:18:04.245081Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders.ArXiv, abs/2408.15998, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19474","last_updated":"2025-05-26T03:53:00Z","snapshot_observed_at":"2026-08-16T13:46:30.825615Z","submitted_at":"2025-05-26T03:53:00Z","title":"Causal-LLaVA: Causal Disentanglement for Mitigating Hallucination in Multimodal Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:18:04.245081Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2505.19474"},"observation_digest":"sha256:8d6f277f84024976275627b78714953197d81f389b321f87430ddb05d7735374","observation_id":"3bbb0d25-af3a-423d-8185-218fc55a05ac","resolution":{"observed_at":"2026-08-07T14:18:04.245081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T13:40:23.941064Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21389","last_updated":"2025-05-27T16:17:15Z","snapshot_observed_at":"2026-08-15T07:49:04.403413Z","submitted_at":"2025-05-27T16:17:15Z","title":"AutoJudger: An Agent-Driven Framework for Efficient Benchmarking of MLLMs","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T13:40:23.941064Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2505.21389"},"observation_digest":"sha256:21c19f2f0bb4d330489e12744a18053f5f350229e8d609854be6773ce045e1aa","observation_id":"d47bf71d-1318-4c93-a5c0-783ff35a655a","resolution":{"observed_at":"2026-08-07T13:40:23.941064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T12:25:08.007033Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24541","last_updated":"2025-05-30T12:48:07Z","snapshot_observed_at":"2026-08-19T18:47:36.846356Z","submitted_at":"2025-05-30T12:48:07Z","title":"Mixpert: Mitigating Multimodal Learning Conflicts with Efficient Mixture-of-Vision-Experts","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T12:25:08.007033Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2505.24541"},"observation_digest":"sha256:fd462798f37946ae0f3063433f3ee118cece68e903560400467a7b91f7f96d49","observation_id":"0c418386-4dd9-4852-b58d-4d8fd9c8d79b","resolution":{"observed_at":"2026-08-07T12:25:08.007033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T11:56:14.998714Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01064","last_updated":"2025-09-14T09:51:48Z","snapshot_observed_at":"2026-08-19T08:35:23.540315Z","submitted_at":"2025-06-01T16:07:30Z","title":"Fighting Fire with Fire (F3): A Training-free and Efficient Visual Adversarial Example Purification Method in LVLMs","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:14.998714Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2506.01064"},"observation_digest":"sha256:f49a0f8900b59a4d68ade5273f60b495b9a0e1a1d519734c903132be7e6084a6","observation_id":"6dc75d72-0f11-49d1-866f-586263a125d2","resolution":{"observed_at":"2026-08-07T11:56:14.998714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T11:26:14.486555Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02557","last_updated":"2025-06-03T07:44:43Z","snapshot_observed_at":"2026-08-12T03:14:23.179364Z","submitted_at":"2025-06-03T07:44:43Z","title":"Kernel-based Unsupervised Embedding Alignment for Enhanced Visual Representation in Vision-language Models","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:14.486555Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2506.02557"},"observation_digest":"sha256:f0dffced88bb654c4be8991f438a480268835b44082b94cf94c278c554b1d380","observation_id":"5efff50b-5682-40c5-b5d9-0ed382beb549","resolution":{"observed_at":"2026-08-07T11:26:14.486555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T05:25:31.936899Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08008","last_updated":"2025-06-09T17:59:54Z","snapshot_observed_at":"2026-08-15T07:02:37.929930Z","submitted_at":"2025-06-09T17:59:54Z","title":"Hidden in plain sight: VLMs overlook their visual representations","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T05:25:31.936899Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2506.08008"},"observation_digest":"sha256:0bb17442068a60f681938893b6471209bc8632ec50f1906f8c16bf74d315745b","observation_id":"1cb2a407-73a5-493f-bfe7-de063c93e6c1","resolution":{"observed_at":"2026-08-07T05:25:31.936899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-07T04:08:49.320477Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11515","last_updated":"2025-06-13T07:16:41Z","snapshot_observed_at":"2026-08-14T02:40:24.668658Z","submitted_at":"2025-06-13T07:16:41Z","title":"Manager: Aggregating Insights from Unimodal Experts in Two-Tower VLMs and MLLMs","version":1},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-08-07T04:08:49.320477Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2506.11515"},"observation_digest":"sha256:8eeebe37ff4453fa29083af45ca203f0f691ea78758f4ea6845a912f05230012","observation_id":"87a83c05-843f-4c94-8eb7-9707e641f268","resolution":{"observed_at":"2026-08-07T04:08:49.320477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-06T23:57:25.522410Z","title":"arXiv preprint arXiv:2408.15998 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15681","last_updated":"2026-06-25T14:33:27Z","snapshot_observed_at":"2026-08-08T08:54:57.204006Z","submitted_at":"2025-06-18T17:59:49Z","title":"GenRecal: Generation after Recalibration from Large to Small Vision-Language Models","version":4},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T23:57:25.522410Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2506.15681"},"observation_digest":"sha256:9b82073c5dbba6874d6ed2cdd63156603c8a5e040532313d99452b84fb883659","observation_id":"5a1ddb53-fd2a-4b2f-977d-72783eacc4ae","resolution":{"observed_at":"2026-08-06T23:57:25.522410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-06T21:19:45.062070Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00505","last_updated":"2025-07-04T13:15:34Z","snapshot_observed_at":"2026-08-15T20:28:40.863571Z","submitted_at":"2025-07-01T07:20:11Z","title":"LLaVA-SP: Enhancing Visual Representation with Visual Spatial Tokens for MLLMs","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T21:19:45.062070Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2507.00505"},"observation_digest":"sha256:59d748ff2fa1d20849f65b40f75292e7ee6b28be977dc6ad4282cd51d4cac91c","observation_id":"0a7bcfc6-44a7-4e98-ac0f-e6e35e9920ad","resolution":{"observed_at":"2026-08-06T21:19:45.062070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-06T17:39:41.625414Z","title":"Eagle: Ex- ploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10300","last_updated":"2025-07-14T14:04:14Z","snapshot_observed_at":"2026-08-15T02:10:47.953200Z","submitted_at":"2025-07-14T14:04:14Z","title":"FaceLLM: A Multimodal Large Language Model for Face Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:39:41.625414Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2507.10300"},"observation_digest":"sha256:fe9f9c1f99ac3fc7a64bb21fa9e7a705a3acec3cb0a9daea6d95d344f0720f31","observation_id":"173447f8-35ee-440e-b605-14493b863059","resolution":{"observed_at":"2026-08-06T17:39:41.625414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-06T15:57:04.082371Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-17T21:12:15.287539Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:04.082371Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:72b53f9946e5a45634a725ad2defd5cfdd99433e0343f412d46ee00ec6a40a7c","observation_id":"a9b76c5c-bc67-46d1-bccd-02a02a98b93d","resolution":{"observed_at":"2026-08-06T15:57:04.082371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2507.16815","last_updated":"2025-09-18T16:26:53Z","snapshot_observed_at":"2026-08-03T17:16:24.219569Z","submitted_at":"2025-07-22T17:59:46Z","title":"ThinkAct: Vision-Language-Action Reasoning via Reinforced Visual Latent Planning","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-19T03:18:14.655384Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2507.16815"},"observation_digest":"sha256:23b5211df6d604c8a8e5a3bbd32eaac7e01024e13ed9e79ae4bdcd8e47452628","observation_id":"69540dfa-b95c-4284-99f5-f8b6e9bd2e0d","resolution":{"observed_at":"2026-05-19T03:22:01.064010Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-06T05:36:20.951429Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01540","last_updated":"2025-08-03T01:49:08Z","snapshot_observed_at":"2026-08-15T08:02:02.409123Z","submitted_at":"2025-08-03T01:49:08Z","title":"MagicVL-2B: Empowering Vision-Language Models on Mobile Devices with Lightweight Visual Encoders via Curriculum Learning","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T05:36:20.951429Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2508.01540"},"observation_digest":"sha256:31918c257a45386ebd3e2211a99b96b6d68104e2ca083cfe7a333f6298566baf","observation_id":"bcf31caa-61fe-4cf8-8152-0b803eaec2a8","resolution":{"observed_at":"2026-08-06T05:36:20.951429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2603.09465","last_updated":"2026-05-11T07:51:23Z","snapshot_observed_at":"2026-08-14T11:17:42.797216Z","submitted_at":"2026-03-10T10:19:07Z","title":"EvoDriveVLA: Evolving Driving VLA Models via Collaborative Perception-Planning Distillation","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T14:14:45.982490Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2603.09465"},"observation_digest":"sha256:f16b42b1b9ef1ae30182cdb0775fbe314d2c965c795d89320c2becb67ad7ee65","observation_id":"acbad68a-4833-4661-9087-b974a4f3c936","resolution":{"observed_at":"2026-05-15T14:15:54.645089Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2604.03231","last_updated":"2026-04-03T17:59:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T17:59:51Z","title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-13T20:28:30.864143Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2604.03231"},"observation_digest":"sha256:c164497ae5653d4dc38e5aac6ea2c6daea4eb8ad2cac560ea242d54138e2e1a6","observation_id":"e3a8aeeb-1a3d-4112-bac7-8ad303b07b30","resolution":{"observed_at":"2026-05-13T20:33:17.129719Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2604.12966","last_updated":"2026-04-14T16:59:53Z","snapshot_observed_at":"2026-08-16T10:59:57.768293Z","submitted_at":"2026-04-14T16:59:53Z","title":"Boosting Visual Instruction Tuning with Self-Supervised Guidance","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-10T16:27:52.208827Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2604.12966"},"observation_digest":"sha256:6b6dc094de79a33002ccf1cb344be2d97d0999b8755789bdf5ca30e589c3c5a6","observation_id":"37faf7fb-2289-46ff-93b1-5f5bedc8cb05","resolution":{"observed_at":"2026-05-11T08:50:58.631506Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2605.11405","last_updated":"2026-05-13T01:55:26Z","snapshot_observed_at":"2026-08-17T07:51:05.413011Z","submitted_at":"2026-05-12T01:51:03Z","title":"20/20 Vision Language Models: A Prescription for Better VLMs through Data Curation Alone","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T02:52:43.674969Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2605.11405"},"observation_digest":"sha256:4e6c73754015719f73ff5ce8b0d478a93b07045675bc2d07886d3f3e575a3af3","observation_id":"3237e382-857d-4ef3-8222-734daab5f285","resolution":{"observed_at":"2026-05-13T02:57:09.535118Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2605.11405","last_updated":"2026-05-13T01:55:26Z","snapshot_observed_at":"2026-08-17T07:51:05.413011Z","submitted_at":"2026-05-12T01:51:03Z","title":"20/20 Vision Language Models: A Prescription for Better VLMs through Data Curation Alone","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-14T21:28:37.680681Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2605.11405"},"observation_digest":"sha256:5d9a918ec47916b49555739f9e966d7561ae8582a433ddeeb8924d5345583b90","observation_id":"adcf702e-8f17-4fd1-8192-70520c6f4469","resolution":{"observed_at":"2026-05-14T21:29:28.665397Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2605.24675","last_updated":"2026-05-23T17:25:45Z","snapshot_observed_at":"2026-08-14T12:09:01.789331Z","submitted_at":"2026-05-23T17:25:45Z","title":"VaaWIT: Visual-Aware Adaptation of Large Language Models for Multilingual Web Image Translation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-30T13:33:12.116833Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2605.24675"},"observation_digest":"sha256:64824be4309c90fe7886ef6b1b33cfc37f1691f7e0416a08a0e3a40d1101b897","observation_id":"5e875590-fa29-4b24-bf74-c5fca19e609c","resolution":{"observed_at":"2026-06-30T13:34:40.251717Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2605.27959","last_updated":"2026-05-28T03:13:45Z","snapshot_observed_at":"2026-08-14T12:43:34.377960Z","submitted_at":"2026-05-27T04:52:42Z","title":"ROVER: Routing Object-Centric Visual Evidence for Grounded Multi-Image Reasoning","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-29T13:41:44.049230Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2605.27959"},"observation_digest":"sha256:99dbc08145e44685c88c01d113d433ebb4dd12d1a0ee97768cb93c6481e329f9","observation_id":"50c94100-c9d2-423f-bdaf-e5f455f52f48","resolution":{"observed_at":"2026-06-29T13:43:28.581895Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2606.03444","last_updated":"2026-06-02T10:28:32Z","snapshot_observed_at":"2026-08-13T23:50:16.854906Z","submitted_at":"2026-06-02T10:28:32Z","title":"PRISM: Synergizing Vision Foundation Models via Self-organized Expert Specialization","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T10:47:33.591670Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2606.03444"},"observation_digest":"sha256:23ec12e4289b743b702b0824691d2dfa6b4aa6e268eebaf91f9df8e44c772286","observation_id":"13c63f98-43b8-4123-8087-67392edd9ea7","resolution":{"observed_at":"2026-07-02T02:36:27.234746Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2606.03713","last_updated":"2026-08-11T09:33:56Z","snapshot_observed_at":"2026-08-14T23:09:30.355546Z","submitted_at":"2026-06-02T14:34:48Z","title":"Investigating Adversarial Robustness of Multi-modal Large Language Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-28T11:11:34.152223Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2606.03713"},"observation_digest":"sha256:5c60b1faaf5ac26936c166ff69b1af0f382ab8221bceea12c7934123cade81cd","observation_id":"57aaec14-49e6-4d60-b199-643e31120ecc","resolution":{"observed_at":"2026-07-02T02:06:27.647252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2606.03879","last_updated":"2026-06-02T16:46:42Z","snapshot_observed_at":"2026-08-07T17:02:56.856514Z","submitted_at":"2026-06-02T16:46:42Z","title":"Beyond Encoder Accumulation: Measuring Encoder Roles in Multi-Encoder VLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T11:08:36.451147Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2606.03879"},"observation_digest":"sha256:dccf17063f3d25148d4899a3647c55c8f7fc17ae5dc99487f749501b2f19e09c","observation_id":"48c98fa9-8ee2-44fc-b467-2fafe97a36fc","resolution":{"observed_at":"2026-07-02T02:16:26.503741Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":"2408.15998","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-07-04T15:09:55.297801Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","venue":null,"work_id":"0fbfe096-026b-4d72-a17c-0d68727084ff","year":2024},"citing_paper":{"arxiv_id":"2606.26196","last_updated":"2026-06-24T15:20:32Z","snapshot_observed_at":"2026-08-17T15:34:17.600386Z","submitted_at":"2026-06-24T15:20:32Z","title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-26T01:50:54.242508Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2606.26196"},"observation_digest":"sha256:93938a4a0e589eab7b342176a7d2fb4a8ed8f0ff28fd07b9de74e70c442c0a22","observation_id":"b268367e-8616-40e0-99ff-af7dc2c90b08","resolution":{"observed_at":"2026-07-04T15:09:55.300051Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-01T21:12:07.027866Z","title":"Eagle: Exploring the design space for multimodal llms with mixture of encoders.arXiv preprint arXiv:2408.15998, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16165","last_updated":"2026-07-17T17:46:23Z","snapshot_observed_at":"2026-08-19T12:48:08.953520Z","submitted_at":"2026-07-17T17:46:23Z","title":"An Exam for Active Observers","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T21:12:07.027866Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2607.16165"},"observation_digest":"sha256:ece80085e240769c13a533b460f88969b28bb3f890e7b18129d33070f56ff36e","observation_id":"90d63a20-dd51-42f5-b4e0-0d5489f55b7b","resolution":{"observed_at":"2026-08-01T21:12:07.027866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15998","snapshot_observed_at":"2026-08-15T15:10:13.173431Z","title":"2025.Ea- gle: Exploring the design space for multimodal llms with mixture of encoders","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.01635","last_updated":"2026-08-03T03:07:54Z","snapshot_observed_at":"2026-08-18T02:02:56.674864Z","submitted_at":"2026-08-03T03:07:54Z","title":"Mitigating Visual Degradation in MLLMs via Spatial-Spectral Visual Anchor Learning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T15:10:13.173431Z"},"links":{"cited_paper":"/paper/2408.15998","citing_paper":"/paper/2608.01635"},"observation_digest":"sha256:b50aad1db4bb5a2fd7af85f097bd885dc3251e5bc5e63173e74cda207396b4c8","observation_id":"49abcba5-801d-4ff5-a088-ac89af31c9ab","resolution":{"observed_at":"2026-08-15T15:10:13.173431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.15998/citation-record","integrity":"/paper/2408.15998/integrity","json":"/paper/2408.15998/citation-record.json","paper":"/paper/2408.15998"},"outbound":[],"paper":{"arxiv_id":"2408.15998","last_updated":"2025-03-02T23:41:37Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T22:32:42.042876Z","submitted_at":"2024-08-28T17:59:31Z","title":"Eagle: Exploring The Design Space for Multimodal LLMs with Mixture of Encoders"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 56 inbound Pith citation observations for arXiv:2408.15998."}