{"as_of":"2026-08-18T00:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e337fc5ec4a09b92609623388ea5f3057d9eaff8d08b3607ef8cfbbd7cdcafd0","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":50,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:12:00.710731Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T19:50:10.187396Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T20:25:33.854923Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2306.13394"},"observation_digest":"sha256:f4218ec1b9c588a7d424f9dfc3e933cc437487112b798e43b747811ea4e17f0f","observation_id":"f07e53ec-0581-49c0-b1d4-c97c9b444c5e","resolution":{"observed_at":"2026-05-10T20:25:34.300055Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-08-17T14:16:52.244007Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":129,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:c636cfceaeb3260d4dd23a2d4be2fedec5e227312b983511900c784ea943c460","observation_id":"2203c694-e3c4-49ff-9ae5-f353746bb08b","resolution":{"observed_at":"2026-05-12T20:58:59.282180Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2405.19088","last_updated":"2026-04-15T02:26:56Z","snapshot_observed_at":"2026-07-06T18:21:57.091728Z","submitted_at":"2024-05-29T13:51:43Z","title":"Cracking the Code of Juxtaposition: Can AI Models Understand the Humorous Contradictions","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-24T00:52:52.056076Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2405.19088"},"observation_digest":"sha256:88ffb3a83c186bc86260d4b832eb7dc6713d5431705f7b3a1ded4f5b833ae473","observation_id":"8cee43a7-b33a-4385-a8e3-8b2429966526","resolution":{"observed_at":"2026-05-24T00:53:40.596039Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2406.09411","last_updated":"2024-07-02T01:56:14Z","snapshot_observed_at":"2026-08-17T21:30:24.278971Z","submitted_at":"2024-06-13T17:59:52Z","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-17T01:09:30.360275Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2406.09411"},"observation_digest":"sha256:251730588ee5a3b091c903ad71656483d25d93af6a8f6e6b8d525c7ac7f32985","observation_id":"1cf8737c-5e06-4ea3-a4af-69cf81b378a6","resolution":{"observed_at":"2026-05-17T01:09:30.450662Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-12T14:31:36.790918Z","title":"Mmt-bench: A comprehen- sive multimodal benchmark for evaluating large vision-language models towards multitask agi,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-14T10:19:04.189589Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T14:31:36.790918Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2411.15296"},"observation_digest":"sha256:591703b229d4bb79d331a6dfb60dc1730a21ec1302fbbbba020ce34ee9c3cfbd","observation_id":"64c8d833-788c-4e0b-8f5d-9d19661b3933","resolution":{"observed_at":"2026-08-12T14:31:36.790918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-12T11:10:54.885006Z","title":"MMT-Bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.18499","last_updated":"2025-03-30T07:22:46Z","snapshot_observed_at":"2026-08-17T14:00:00.020812Z","submitted_at":"2024-11-27T16:39:04Z","title":"OpenING: A Comprehensive Benchmark for Judging Open-ended Interleaved Image-Text Generation","version":3},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-12T11:10:54.885006Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2411.18499"},"observation_digest":"sha256:fcea096c494adc8e599bded1a685afeeab5abcefa29233892563c5915ca15625","observation_id":"2fb0e1f1-6f62-4fc8-846b-6f1d09200c36","resolution":{"observed_at":"2026-08-12T11:10:54.885006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-11T21:30:00.204585Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.04447","last_updated":"2025-04-11T07:10:02Z","snapshot_observed_at":"2026-08-14T15:38:54.058153Z","submitted_at":"2024-12-05T18:57:23Z","title":"EgoPlan-Bench2: A Benchmark for Multimodal Large Language Model Planning in Real-World Scenarios","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T21:30:00.204585Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2412.04447"},"observation_digest":"sha256:b8d6dc4f1d056752a4fe401afd5ec5cb2ceb874896327fe0ee8ff8dbedab57e8","observation_id":"2b1ed913-0ec8-46fb-bc60-0014c828bf2a","resolution":{"observed_at":"2026-08-11T21:30:00.204585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":278,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:ea91327db6e761c95f7faea01dec1af1d1ffdf1df0371c9aada2ecd215a240b6","observation_id":"41f9a92b-b410-4268-9701-a2246ee958eb","resolution":{"observed_at":"2026-05-10T13:23:58.241936Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-11T13:58:43.757280Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12606","last_updated":"2024-12-17T07:06:10Z","snapshot_observed_at":"2026-08-14T23:19:29.181702Z","submitted_at":"2024-12-17T07:06:10Z","title":"Multi-Dimensional Insights: Benchmarking Real-World Personalization in Large Multimodal Models","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-11T13:58:43.757280Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2412.12606"},"observation_digest":"sha256:1e629e850ec931a7a454fc04f996d03f287a36c28114a7fba806fd210677962a","observation_id":"3f4071c6-3d95-48d8-91fc-6fcd2733ea34","resolution":{"observed_at":"2026-08-11T13:58:43.757280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-12T17:21:52.298102Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:89fe117588f4bb655a508066028569e30a666d4939929fd81965131535c70267","observation_id":"9b89acf0-6d5f-4b6d-9e8a-432c6991a3c8","resolution":{"observed_at":"2026-05-17T20:33:26.733088Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-10T21:18:59.777581Z","title":"Mmt-bench: A comprehensive multimodal bench- mark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05444","last_updated":"2025-01-09T18:55:52Z","snapshot_observed_at":"2026-08-13T08:24:02.836306Z","submitted_at":"2025-01-09T18:55:52Z","title":"Can MLLMs Reason in Multimodality? EMMA: An Enhanced MultiModal ReAsoning Benchmark","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-10T21:18:59.777581Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2501.05444"},"observation_digest":"sha256:8d4c34bd0d9ffe0816963b73fdd9d973abb3da451ba233917f5dda068188d1d0","observation_id":"94ee971f-2195-4d7b-ab91-d332cb320c83","resolution":{"observed_at":"2026-08-10T21:18:59.777581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-10T18:53:21.313329Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.10967","last_updated":"2025-02-12T08:10:06Z","snapshot_observed_at":"2026-08-13T22:41:42.954852Z","submitted_at":"2025-01-19T07:00:46Z","title":"Advancing General Multimodal Capability of Vision-language Models with Pyramid-descent Visual Position Encoding","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-10T18:53:21.313329Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2501.10967"},"observation_digest":"sha256:4772905c53319925a225e5c46b34fbb3d51c65e7c1156ebecc6eca006afef12b","observation_id":"90c41d2a-8e4a-4132-bf31-981baf459a15","resolution":{"observed_at":"2026-08-10T18:53:21.313329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-07T22:43:59.488259Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09093","last_updated":"2025-02-13T09:04:28Z","snapshot_observed_at":"2026-08-16T12:52:38.093535Z","submitted_at":"2025-02-13T09:04:28Z","title":"From Visuals to Vocabulary: Establishing Equivalence Between Image and Text Token Through Autoregressive Pre-training in MLLMs","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T22:43:59.488259Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2502.09093"},"observation_digest":"sha256:2f35bb723c5d58263ba23b150bb8c5ade08928772263d6ad647858a2d14ba5a2","observation_id":"0e128430-72d1-4109-8572-04960d51c3ba","resolution":{"observed_at":"2026-08-07T22:43:59.488259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2503.23137","last_updated":"2026-04-15T02:38:45Z","snapshot_observed_at":"2026-08-12T16:01:25.076287Z","submitted_at":"2025-03-29T16:08:51Z","title":"When 'YES' Meets 'BUT': Can Large Models Comprehend Contradictory Humor Through Comparative Reasoning?","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-22T22:38:35.969273Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2503.23137"},"observation_digest":"sha256:cd55391772676c2bb289e42a6213fb5697b3c707051309abf2b948490b37927f","observation_id":"122975f6-eed8-4776-8518-f9f0845635f9","resolution":{"observed_at":"2026-05-22T22:42:13.621900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-17T09:56:52.502317Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":138,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:eba37b40089b4a4fdacda45b383fd8dc814d8a755963722942c0b75328ec01a6","observation_id":"61ff7e0a-45e4-481a-840e-ac8901063e05","resolution":{"observed_at":"2026-05-10T13:41:08.153948Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-16T10:12:00.710731Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluat- ing large vision-language models towards multitask agi,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18838","last_updated":"2025-04-26T07:48:52Z","snapshot_observed_at":"2026-08-17T04:06:12.060099Z","submitted_at":"2025-04-26T07:48:52Z","title":"Toward Generalizable Evaluation in the LLM Era: A Survey Beyond Benchmarks","version":1},"reference_index":156,"source":"pdf_text","source_observed_at":"2026-08-16T10:12:00.710731Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2504.18838"},"observation_digest":"sha256:5b0579dfbd7c4d4131096ad9afbecb6f02a2fa1e4f4d7bb60faaa12f57c94f4b","observation_id":"3b940da0-3cc7-4cf0-a4d7-ebc30e705242","resolution":{"observed_at":"2026-08-16T10:12:00.710731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-15T21:07:59.809431Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10917","last_updated":"2025-05-19T09:40:28Z","snapshot_observed_at":"2026-08-17T22:48:41.335447Z","submitted_at":"2025-05-16T06:43:02Z","title":"VISTA: Enhancing Vision-Text Alignment in MLLMs via Cross-Modal Mutual Information Maximization","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T21:07:59.809431Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2505.10917"},"observation_digest":"sha256:6e3de2591462a7a6d05c34f8d0b0bdd2d28fb19c3f9350f9098bdebbf188e6b7","observation_id":"67fd94ee-9da5-4dbd-b54e-bf48ab174532","resolution":{"observed_at":"2026-08-15T21:07:59.809431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-15T20:44:43.699366Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.12207","last_updated":"2025-08-13T05:17:53Z","snapshot_observed_at":"2026-08-17T12:05:24.394796Z","submitted_at":"2025-05-18T02:45:19Z","title":"Can Large Multimodal Models Understand Agricultural Scenes? Benchmarking with AgroMind","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T20:44:43.699366Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2505.12207"},"observation_digest":"sha256:f821adb8fde7a3ec6bb55d1ee42221c393accf9cbae0dc2243c6149f55f0a330","observation_id":"05d5a28d-78ef-46e4-87e7-9f579b2fc602","resolution":{"observed_at":"2026-08-15T20:44:43.699366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-07T13:40:25.406739Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21389","last_updated":"2025-05-27T16:17:15Z","snapshot_observed_at":"2026-08-15T07:49:04.403413Z","submitted_at":"2025-05-27T16:17:15Z","title":"AutoJudger: An Agent-Driven Framework for Efficient Benchmarking of MLLMs","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T13:40:25.406739Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2505.21389"},"observation_digest":"sha256:d55cfd0e49e1ed0c2c6bfab8c04dfb6eec96ba33d0a4f39eb927a8177bd559fa","observation_id":"dd1247fd-ca59-4c6e-8d84-ab8666766be7","resolution":{"observed_at":"2026-08-07T13:40:25.406739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-07T10:52:35.642571Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04141","last_updated":"2026-07-20T11:08:55Z","snapshot_observed_at":"2026-08-15T06:32:42.325375Z","submitted_at":"2025-06-04T16:33:41Z","title":"MMR-V: What's Left Unsaid? A Benchmark for Multimodal Deep Reasoning in Videos","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:52:35.642571Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2506.04141"},"observation_digest":"sha256:1745f2ee0f2d18c95b4f9fba1854774e1491cda660c045d95d3d53e12a4cdf22","observation_id":"0552af4a-974a-47da-b9ed-a3540cec5b83","resolution":{"observed_at":"2026-08-07T10:52:35.642571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-07T06:02:30.488263Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06279","last_updated":"2025-06-06T17:59:06Z","snapshot_observed_at":"2026-08-15T00:15:02.626802Z","submitted_at":"2025-06-06T17:59:06Z","title":"CoMemo: LVLMs Need Image Context with Image Memory","version":1},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-08-07T06:02:30.488263Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2506.06279"},"observation_digest":"sha256:1fc48f35bd3396c84fe9a9e4600fb04452658e5b82563f0eb19d377ae71f1a73","observation_id":"359cf8d7-ad72-45e5-8231-774b361010e1","resolution":{"observed_at":"2026-08-07T06:02:30.488263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-07T11:19:05.062097Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14805","last_updated":"2025-08-12T17:07:53Z","snapshot_observed_at":"2026-08-15T10:53:20.402399Z","submitted_at":"2025-06-03T13:44:14Z","title":"Argus Inspection: Do Multimodal Large Language Models Possess the Eye of Panoptes?","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T11:19:05.062097Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2506.14805"},"observation_digest":"sha256:e9498c5621445d5dbf17e9c33c059a1ba956542475dcf37e93088cd99dbbd673","observation_id":"b975bd57-1ac1-4c2c-9a4b-a220e9bdc4d1","resolution":{"observed_at":"2026-08-07T11:19:05.062097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-15T18:52:30.947865Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18378","last_updated":"2025-06-23T08:11:24Z","snapshot_observed_at":"2026-08-17T13:52:55.913164Z","submitted_at":"2025-06-23T08:11:24Z","title":"Taming Vision-Language Models for Medical Image Analysis: A Comprehensive Review","version":1},"reference_index":231,"source":"pdf_text","source_observed_at":"2026-08-15T18:52:30.947865Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2506.18378"},"observation_digest":"sha256:80e2ce6328cdf4c16ccbe0e6f69c7105fcb956fee007ae3b08a38cead0609ed2","observation_id":"e302ac2a-9625-464c-9a83-d3a02c2d75d8","resolution":{"observed_at":"2026-08-15T18:52:30.947865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-06T21:41:40.493129Z","title":"Mmt-bench: A comprehensive multimodal bench- mark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23563","last_updated":"2025-06-30T07:14:38Z","snapshot_observed_at":"2026-08-12T15:29:00.237188Z","submitted_at":"2025-06-30T07:14:38Z","title":"MMReason: An Open-Ended Multi-Modal Multi-Step Reasoning Benchmark for MLLMs Toward AGI","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T21:41:40.493129Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2506.23563"},"observation_digest":"sha256:6384c4cc60afab88c143e93501e202168303b50dea5c59244bda29b95cd0eec2","observation_id":"9ea9b439-fd49-40a5-9db9-5e1f2a8f0b91","resolution":{"observed_at":"2026-08-06T21:41:40.493129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-06T18:19:59.092552Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluatinglargevision-languagemodelstowardsmultitaskagi","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08575","last_updated":"2025-07-11T13:23:25Z","snapshot_observed_at":"2026-08-09T01:20:22.458672Z","submitted_at":"2025-07-11T13:23:25Z","title":"Large Multi-modal Model Cartographic Map Comprehension for Textual Locality Georeferencing","version":1},"reference_index":2007,"source":"pdf_text","source_observed_at":"2026-08-06T18:19:59.092552Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2507.08575"},"observation_digest":"sha256:14ab6f1a5c9edec2aefa24ab3c5ecc6e243c06615feaca9474bdf741e768c30f","observation_id":"194404aa-8c2c-4111-bd6d-595658e045ed","resolution":{"observed_at":"2026-08-06T18:19:59.092552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-06T17:49:34.927360Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-13T20:26:55.109929Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.927360Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:0b1979c6a71802eac93730a799d9389fc4bb7140bbe6be84d2be2348342bb970","observation_id":"5dfb38a4-4956-47d4-9875-0bea4f982ff9","resolution":{"observed_at":"2026-08-06T17:49:34.927360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-06T16:41:13.357316Z","title":"Mmt- bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13405","last_updated":"2025-07-17T04:47:47Z","snapshot_observed_at":"2026-08-13T10:22:21.170507Z","submitted_at":"2025-07-17T04:47:47Z","title":"COREVQA: A Crowd Observation and Reasoning Entailment Visual Question Answering Benchmark","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T16:41:13.357316Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2507.13405"},"observation_digest":"sha256:9ed959ad745c6eff2caf40de33c135d65fc39bb994085f2ede02d6bc2509712b","observation_id":"49b1b18b-a872-47ff-8655-2e2f8a63c8c1","resolution":{"observed_at":"2026-08-06T16:41:13.357316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-06T10:53:20.657760Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi.arXiv preprint arXiv:2404.16006, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.23382","last_updated":"2025-07-31T09:59:17Z","snapshot_observed_at":"2026-08-07T11:02:57.710023Z","submitted_at":"2025-07-31T09:59:17Z","title":"MPCC: A Novel Benchmark for Multimodal Planning with Complex Constraints in Multimodal Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T10:53:20.657760Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2507.23382"},"observation_digest":"sha256:7fc0233e45e75198b0ac66a797a7f2535b225970edefbe755c6ae1691996a5a0","observation_id":"26a3a3dc-d9ba-4060-ba7c-8bd3e211a70c","resolution":{"observed_at":"2026-08-06T10:53:20.657760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2508.05748","last_updated":"2025-09-01T03:21:53Z","snapshot_observed_at":"2026-08-12T15:21:22.765600Z","submitted_at":"2025-08-07T18:03:50Z","title":"WebWatcher: Breaking New Frontier of Vision-Language Deep Research Agent","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-15T18:56:23.817544Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2508.05748"},"observation_digest":"sha256:42b6b94597e4f1a7a2d8098c6a741cd661bd1877064d200d03cc80dc5b70d513","observation_id":"623233a9-efc2-4d9b-826d-7f857dfd8e81","resolution":{"observed_at":"2026-05-15T18:56:24.034271Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-05T19:03:05.388212Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.13692","last_updated":"2025-08-19T09:52:04Z","snapshot_observed_at":"2026-08-16T10:29:42.442699Z","submitted_at":"2025-08-19T09:52:04Z","title":"HumanPCR: Probing MLLM Capabilities in Diverse Human-Centric Scenes","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T19:03:05.388212Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2508.13692"},"observation_digest":"sha256:613e4ca058e1869ca1c8b956a9023f5c06d6c24218577f094fd526a346c52c1d","observation_id":"7b3eee89-68d3-4e4b-8a3b-26b972ad55da","resolution":{"observed_at":"2026-08-05T19:03:05.388212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-05T16:32:43.360928Z","title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.18179","last_updated":"2025-08-25T16:33:07Z","snapshot_observed_at":"2026-08-13T11:32:51.771701Z","submitted_at":"2025-08-25T16:33:07Z","title":"SEAM: Semantically Equivalent Across Modalities Benchmark for Vision-Language Models","version":1},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-08-05T16:32:43.360928Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2508.18179"},"observation_digest":"sha256:3c856ea715ec5f60f0a209b4c743cabfd56c09b7ecfe609494c9eec992c0ede0","observation_id":"6cfdf7c3-1f09-4e39-8537-ae6415cfb2a5","resolution":{"observed_at":"2026-08-05T16:32:43.360928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-08-17T12:32:16.575866Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":167,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:9a4e6013b3c02e3d545b32666cbca53783f4d30e48fa536e20108a6d9ecd9dee","observation_id":"9d72ab86-6f58-411f-83cb-727847e57973","resolution":{"observed_at":"2026-05-10T11:58:58.921223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-04T19:27:40.845414Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09254","last_updated":"2025-09-11T08:39:08Z","snapshot_observed_at":"2026-08-16T17:44:52.264664Z","submitted_at":"2025-09-11T08:39:08Z","title":"Towards Better Dental AI: A Multimodal Benchmark and Instruction Dataset for Panoramic X-ray Analysis","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-04T19:27:40.845414Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2509.09254"},"observation_digest":"sha256:75c86bbb6d0ae32fa87080b28fe57601b8aadbf15fa82708c38019b9abed84ad","observation_id":"d08a92f7-3106-4809-bb68-07ac28c4ed8c","resolution":{"observed_at":"2026-08-04T19:27:40.845414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2511.20814","last_updated":"2026-04-05T23:59:08Z","snapshot_observed_at":"2026-08-06T09:15:03.837609Z","submitted_at":"2025-11-25T20:00:47Z","title":"SPHINX: A Synthetic Environment for Visual Perception and Reasoning","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-17T04:19:26.808804Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2511.20814"},"observation_digest":"sha256:0be4b3c2bc2b5e5dacdbfaabfd74a75f6e5a829984ded18c10cbd6a7ad89ee29","observation_id":"1bde5717-df73-4b65-8f18-d240ca650786","resolution":{"observed_at":"2026-05-17T04:21:30.726988Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2512.03043","last_updated":"2026-04-28T12:07:36Z","snapshot_observed_at":"2026-08-13T15:37:47.305433Z","submitted_at":"2025-12-02T18:59:52Z","title":"OneThinker: All-in-one Reasoning Model for Image and Video","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T02:09:39.820651Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2512.03043"},"observation_digest":"sha256:82b67d37e8597fbc756444b198553642b4a4081443b989aa99b435df0855ee12","observation_id":"a7e25596-8751-4631-ad05-a1ae6b969843","resolution":{"observed_at":"2026-05-17T02:11:26.465826Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2512.03438","last_updated":"2026-07-30T22:27:17Z","snapshot_observed_at":"2026-08-12T16:45:42.496889Z","submitted_at":"2025-12-03T04:42:47Z","title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-17T03:09:31.161760Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2512.03438"},"observation_digest":"sha256:6d13fd9a0f8f4ee6d42cb26238e856db9584dcbf97d2b610847f6b3455cb7879","observation_id":"1e326ccb-4340-43a7-99de-9e432e5af207","resolution":{"observed_at":"2026-05-17T03:11:29.864333Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-03T18:51:48.676604Z","title":"Mmt-bench: A comprehensive multimodal bench- mark for evaluating large vision-language models towards multitask agi.arXiv preprint arXiv:2404.16006, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.03438","last_updated":"2026-07-30T22:27:17Z","snapshot_observed_at":"2026-08-12T16:45:42.496889Z","submitted_at":"2025-12-03T04:42:47Z","title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-03T18:51:48.676604Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2512.03438"},"observation_digest":"sha256:0b303ca1ea3a46353c9023a4ce96a7601734ff95377767aec6385754589a740e","observation_id":"e9f9dc04-1acd-416f-ac6a-8f94a592e8ba","resolution":{"observed_at":"2026-08-03T18:51:48.676604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2604.03893","last_updated":"2026-06-01T03:09:36Z","snapshot_observed_at":"2026-08-12T12:39:42.402287Z","submitted_at":"2026-04-04T23:18:58Z","title":"FeynmanBench: Benchmarking Multimodal LLMs on Diagrammatic Physics Reasoning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-13T16:51:48.705876Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2604.03893"},"observation_digest":"sha256:8e8e488f48b65e7e48096cd126c85a76291408d3991b32b4813a24a97c1bfedf","observation_id":"2a854ec4-9137-4869-94f1-f5370a6ef96f","resolution":{"observed_at":"2026-05-13T16:52:59.949730Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-13T12:10:53.720348Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.03893","last_updated":"2026-06-01T03:09:36Z","snapshot_observed_at":"2026-08-12T12:39:42.402287Z","submitted_at":"2026-04-04T23:18:58Z","title":"FeynmanBench: Benchmarking Multimodal LLMs on Diagrammatic Physics Reasoning","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-07-13T12:10:53.720348Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2604.03893"},"observation_digest":"sha256:8cb8d94bf49c82fa26ad7057f8f76a86936251d6d7dbfc71c369e04b14550834","observation_id":"2f087083-3736-4a58-96f2-e94383d27b47","resolution":{"observed_at":"2026-07-13T12:10:53.720348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2604.08884","last_updated":"2026-04-10T02:47:32Z","snapshot_observed_at":"2026-08-10T22:18:32.055450Z","submitted_at":"2026-04-10T02:47:32Z","title":"HM-Bench: A Comprehensive Benchmark for Multimodal Large Language Models in Hyperspectral Remote Sensing","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T18:06:49.114269Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2604.08884"},"observation_digest":"sha256:35c64fdf2da26cc1375072e6f3cdbbb0ac1b60cf394d7b39dc4c4ed2bba059c5","observation_id":"b62ba793-183d-4028-87e8-a70d8dab483c","resolution":{"observed_at":"2026-05-11T05:30:57.859380Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2605.00814","last_updated":"2026-05-08T16:52:48Z","snapshot_observed_at":"2026-08-13T11:26:37.424755Z","submitted_at":"2026-05-01T17:54:37Z","title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-09T18:53:06.494640Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2605.00814"},"observation_digest":"sha256:c3f1051f329a5e29194c0327a6c623bde6a51d5f6259d02ca5b503604ae50f4d","observation_id":"17f99df8-7dbb-49af-bc06-2fcabbc55bbd","resolution":{"observed_at":"2026-05-11T16:01:22.671101Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2605.00814","last_updated":"2026-05-08T16:52:48Z","snapshot_observed_at":"2026-08-13T11:26:37.424755Z","submitted_at":"2026-05-01T17:54:37Z","title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-11T01:49:15.136031Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2605.00814"},"observation_digest":"sha256:be4989925ec0615b719cfbb7b407c3bfb11c5231dc5b44b61c390a5679a319cd","observation_id":"739c674d-a30c-48f2-8f91-e0179c88281d","resolution":{"observed_at":"2026-05-11T01:50:51.555655Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2605.07593","last_updated":"2026-05-08T11:06:43Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T11:06:43Z","title":"TraceAV-Bench: Benchmarking Multi-Hop Trajectory Reasoning over Long Audio-Visual Videos","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-11T01:53:01.939765Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2605.07593"},"observation_digest":"sha256:b457c6acfaeff8be4813429bf1b7e8ffbdb324670b45e3ba1f25c75ece90b02c","observation_id":"575ae2a5-bd24-453a-abba-a6d0cea6c8a8","resolution":{"observed_at":"2026-05-11T04:15:56.318295Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2606.00535","last_updated":"2026-05-30T05:05:24Z","snapshot_observed_at":"2026-08-16T00:43:44.943332Z","submitted_at":"2026-05-30T05:05:24Z","title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","version":1},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-06-28T18:55:51.474956Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2606.00535"},"observation_digest":"sha256:9ee7fbd747fb03d10ce71e5c5545f44803d6784a8d72b21ff5d4758b5363052b","observation_id":"dac1914c-acbc-439b-b6b9-538dd6b00fbd","resolution":{"observed_at":"2026-06-28T19:02:34.595588Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2606.08464","last_updated":"2026-06-07T05:58:39Z","snapshot_observed_at":"2026-08-15T15:16:57.255291Z","submitted_at":"2026-06-07T05:58:39Z","title":"TVI-CoT: Text-Visual Interleaved Chain-of-Thought Reasoning for Multimodal Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T18:54:34.353940Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2606.08464"},"observation_digest":"sha256:ae50cc0db7b5f297ec5f150abf84c6cfca581e00c754b49e47f7fc54c3a489cd","observation_id":"ccd1fb11-eedc-46fc-a1d7-f1fed4e22ceb","resolution":{"observed_at":"2026-07-02T22:27:25.893702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2606.25445","last_updated":"2026-06-24T06:15:24Z","snapshot_observed_at":"2026-08-13T00:06:24.757720Z","submitted_at":"2026-06-24T06:15:24Z","title":"C3-Bench: A Context-Aware Change Captioning Benchmark","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-06-25T21:02:52.529391Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2606.25445"},"observation_digest":"sha256:aa5591c332ac61044458ca9a86814baca5483750781ab47529040fae268fbe08","observation_id":"03d97ab5-429c-4341-9b82-2c8f44b93dad","resolution":{"observed_at":"2026-07-04T19:50:10.188929Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":"2404.16006","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-04T19:50:10.187396Z","title":"Mmt-bench: A comprehensive multimodal benchmark for eval- uating large vision-language models towards multitask agi","venue":null,"work_id":"b311f9ac-1b41-40b6-b154-df5836a51696","year":2024},"citing_paper":{"arxiv_id":"2606.31986","last_updated":"2026-08-02T06:35:56Z","snapshot_observed_at":"2026-08-15T10:20:43.925593Z","submitted_at":"2026-06-30T17:24:40Z","title":"CoLT: Teaching Multi-Modal Models to Think with Chain of Latent Thoughts","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-07-01T05:41:20.492139Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2606.31986"},"observation_digest":"sha256:889ba5486abaab41d78156dc23cf5fa675d319e7a36cb1afc769bbdf14d834cf","observation_id":"479473d3-2b3b-4790-aa44-1eceee4481a3","resolution":{"observed_at":"2026-07-01T10:15:44.513584Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-07-12T09:59:21.899377Z","title":"arXiv preprint arXiv:2404.16006 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.31986","last_updated":"2026-08-02T06:35:56Z","snapshot_observed_at":"2026-08-15T10:20:43.925593Z","submitted_at":"2026-06-30T17:24:40Z","title":"CoLT: Teaching Multi-Modal Models to Think with Chain of Latent Thoughts","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-07-12T09:59:21.899377Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2606.31986"},"observation_digest":"sha256:2171be4cd33aac42ffd26bf965669c53ad6b4983c85d89d3a5128e9aecb92bc9","observation_id":"9b5d014b-bf73-4425-b7b5-cf1bcdd2f25a","resolution":{"observed_at":"2026-07-12T09:59:21.899377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-04T04:38:23.452380Z","title":"arXiv preprint arXiv:2404.16006 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.31986","last_updated":"2026-08-02T06:35:56Z","snapshot_observed_at":"2026-08-15T10:20:43.925593Z","submitted_at":"2026-06-30T17:24:40Z","title":"CoLT: Teaching Multi-Modal Models to Think with Chain of Latent Thoughts","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-04T04:38:23.452380Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2606.31986"},"observation_digest":"sha256:447ff91cc7841f59a5b95df0bc77049550db7aa98ba0b8467527211450ee4d8b","observation_id":"a36d6b0f-1480-48ef-bb41-49ab19ee5561","resolution":{"observed_at":"2026-08-04T04:38:23.452380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-07T22:11:21.555191Z","title":"arXiv preprint arXiv:2404.16006","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05864","last_updated":"2026-08-06T10:43:33Z","snapshot_observed_at":"2026-08-13T14:54:56.407916Z","submitted_at":"2026-08-06T10:43:33Z","title":"Seeing Is Not Deciding: Can Multimodal LLMs Act as Effective CEOs?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T22:11:21.555191Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2608.05864"},"observation_digest":"sha256:d548e153926aaa60638c359dcd95e85db5b0bad3cba26699e2b8325973cabb0d","observation_id":"51cafaa1-e006-4bd0-a2f0-c01b90c8e58b","resolution":{"observed_at":"2026-08-07T22:11:21.555191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2404.16006/citation-record","integrity":"/paper/2404.16006/integrity","json":"/paper/2404.16006/citation-record.json","paper":"/paper/2404.16006"},"outbound":[],"paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T13:58:08.863436Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 50 inbound Pith citation observations for arXiv:2404.16006."}