{"as_of":"2026-08-11T14:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:85873471efd7b8e072c6ba6b8d446f121ed56df32ddf4cc7007d2fd000fe2492","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T22:08:08.997103Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":4,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-10T22:08:08.997103Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.02765","last_updated":"2025-01-06T05:15:59Z","snapshot_observed_at":"2026-08-11T04:56:13.579583Z","submitted_at":"2025-01-06T05:15:59Z","title":"Visual Large Language Models for Generalized and Specialized Applications","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T22:08:08.997103Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2501.02765"},"observation_digest":"sha256:07220b54a84029f9fcad06f7a240aab02209a191719c5dddcb5c1e56904b8b34","observation_id":"58b0312e-73ff-42f8-b310-a2d89b9933b8","resolution":{"observed_at":"2026-08-10T22:08:08.997103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2502.02871","last_updated":"2026-04-20T02:18:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-05T04:05:27Z","title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-23T04:30:38.804702Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2502.02871"},"observation_digest":"sha256:b13cf6467d717fefb6cce4cdaf790d121d2dc9f4864c1c0672a99c88a570c6f1","observation_id":"d69f840a-ba74-4bf5-b900-dd7928b931eb","resolution":{"observed_at":"2026-05-23T04:32:32.989284Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T18:23:49.849204Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.10391","last_updated":"2025-02-14T18:59:51Z","snapshot_observed_at":"2026-08-08T01:24:46.892879Z","submitted_at":"2025-02-14T18:59:51Z","title":"MM-RLHF: The Next Step Forward in Multimodal LLM Alignment","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T18:23:49.849204Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2502.10391"},"observation_digest":"sha256:467fa00a0e986284edb339fec0868564b1bf5b59be96b90247f9ff1ca8500ae1","observation_id":"e71ccea0-7fe7-42a3-88eb-843910d7a257","resolution":{"observed_at":"2026-08-07T18:23:49.849204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2505.15616","last_updated":"2026-05-13T12:54:20Z","snapshot_observed_at":"2026-08-03T01:38:16.643054Z","submitted_at":"2025-05-21T15:06:59Z","title":"LENS: Multi-level Evaluation of Multimodal Reasoning with Large Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T13:47:51.436258Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2505.15616"},"observation_digest":"sha256:0521ffb77655cf736f8151a664b6de3c11a36f3b21838afcef60e0464da024be","observation_id":"91af4587-f0ff-4525-8ae0-dfbce396dc6f","resolution":{"observed_at":"2026-05-22T13:51:37.685457Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T15:15:41.573252Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15804","last_updated":"2025-07-10T17:36:35Z","snapshot_observed_at":"2026-08-10T13:20:59.425681Z","submitted_at":"2025-05-21T17:57:38Z","title":"STAR-R1: Spatial TrAnsformation Reasoning by Reinforcing Multimodal LLMs","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:15:41.573252Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2505.15804"},"observation_digest":"sha256:86b30d8681b434d61c0a82d06e99bb56f5915ff1bf31050bf444849322b25c7c","observation_id":"5d73b1f5-bc43-45ea-8adc-38778701ecfd","resolution":{"observed_at":"2026-08-07T15:15:41.573252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T12:35:31.362263Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24714","last_updated":"2025-05-30T15:36:19Z","snapshot_observed_at":"2026-08-07T21:26:09.635795Z","submitted_at":"2025-05-30T15:36:19Z","title":"FinMME: Benchmark Dataset for Financial Multi-Modal Reasoning Evaluation","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:31.362263Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2505.24714"},"observation_digest":"sha256:7b2534b1884ffc2da86d52f6d91ce1e4f2282431c64bd286cabad26bbe4ba6af","observation_id":"f31ac3a0-02af-4b1c-aa28-ff6583550664","resolution":{"observed_at":"2026-08-07T12:35:31.362263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T11:51:15.288212Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01293","last_updated":"2025-06-02T04:00:35Z","snapshot_observed_at":"2026-08-07T17:04:28.844476Z","submitted_at":"2025-06-02T04:00:35Z","title":"Abstractive Visual Understanding of Multi-modal Structured Knowledge: A New Perspective for MLLM Evaluation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:15.288212Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2506.01293"},"observation_digest":"sha256:e26bcdfcaac80dbf321bbe8a07a135dd00d6f01a762e32e548e31882cd52c1ba","observation_id":"15910943-7fd6-4bd0-975b-f2f21ac59ccd","resolution":{"observed_at":"2026-08-07T11:51:15.288212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T10:35:40.742944Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05440","last_updated":"2025-06-05T12:43:10Z","snapshot_observed_at":"2026-08-07T21:22:23.284742Z","submitted_at":"2025-06-05T12:43:10Z","title":"BYO-Eval: Build Your Own Dataset for Fine-Grained Visual Assessment of Multimodal Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:35:40.742944Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2506.05440"},"observation_digest":"sha256:4d1164a98a9fa6e7066249211a9fc5876d87c85cd43873c00a2bbdfbfbba73ab","observation_id":"8a236e54-b53d-4350-96ea-4fa5bbd5bdf2","resolution":{"observed_at":"2026-08-07T10:35:40.742944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T05:53:58.992333Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06729","last_updated":"2025-06-07T09:27:26Z","snapshot_observed_at":"2026-08-10T07:02:38.987852Z","submitted_at":"2025-06-07T09:27:26Z","title":"Mitigating Object Hallucination via Robust Local Perception Search","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T05:53:58.992333Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2506.06729"},"observation_digest":"sha256:c915fb584b6bf766ded2bb244ebb0e1f1d1e32cbaa3c43cabe0c79339da329f2","observation_id":"faa2101a-7e86-4794-ae79-813cfdec8a38","resolution":{"observed_at":"2026-08-07T05:53:58.992333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T05:43:40.351303Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms.arXiv preprint arXiv:2411.15296, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07227","last_updated":"2025-06-08T17:23:36Z","snapshot_observed_at":"2026-08-08T06:06:25.367121Z","submitted_at":"2025-06-08T17:23:36Z","title":"Hallucination at a Glance: Controlled Visual Edits and Fine-Grained Multimodal Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:43:40.351303Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2506.07227"},"observation_digest":"sha256:66f37bd6c87677b3effd682940fa0b24168e370f2cd24494adcd98431d457502","observation_id":"ceaae5c7-6498-4941-8dce-05f664adf7a8","resolution":{"observed_at":"2026-08-07T05:43:40.351303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T04:43:42.097080Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09954","last_updated":"2025-06-11T17:23:41Z","snapshot_observed_at":"2026-08-07T21:45:10.885875Z","submitted_at":"2025-06-11T17:23:41Z","title":"Vision Generalist Model: A Survey","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T04:43:42.097080Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2506.09954"},"observation_digest":"sha256:e1917a33d6f717facb87e2be6b8e3f8eb657902d604ff88c36d54657a5521cd0","observation_id":"1198782c-3978-4d7f-bd57-b5177a52c381","resolution":{"observed_at":"2026-08-07T04:43:42.097080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-07T04:07:32.177092Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11571","last_updated":"2025-07-18T08:23:14Z","snapshot_observed_at":"2026-08-08T02:11:07.665188Z","submitted_at":"2025-06-13T08:27:45Z","title":"VFaith: Do Large Multimodal Models Really Reason on Seen Images Rather than Previous Memories?","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T04:07:32.177092Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2506.11571"},"observation_digest":"sha256:282aa732fa124c1dd4410322863580b9c09c3dc79fb9d8bd7deae89a6e0677c9","observation_id":"3a649fa7-a297-4640-82a1-1d40ae8fa644","resolution":{"observed_at":"2026-08-07T04:07:32.177092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-06T16:41:12.474596Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13405","last_updated":"2025-07-17T04:47:47Z","snapshot_observed_at":"2026-08-09T07:59:53.314166Z","submitted_at":"2025-07-17T04:47:47Z","title":"COREVQA: A Crowd Observation and Reasoning Entailment Visual Question Answering Benchmark","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T16:41:12.474596Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2507.13405"},"observation_digest":"sha256:68ce1d81e7d0c663046f9928fa3079f14b04b97ab6a4fd3b835e5c8a972a4a39","observation_id":"b305033a-7bc9-47a0-86da-c716cee83a57","resolution":{"observed_at":"2026-08-06T16:41:12.474596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-06T15:32:34.567976Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15652","last_updated":"2025-07-21T14:15:34Z","snapshot_observed_at":"2026-08-08T07:09:28.862790Z","submitted_at":"2025-07-21T14:15:34Z","title":"Extracting Visual Facts from Intermediate Layers for Mitigating Hallucinations in Multimodal Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T15:32:34.567976Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2507.15652"},"observation_digest":"sha256:fbc99e0c48d3362c2ab3514eaf15370ff171777e9a84dfa35e886f7ca0197cf3","observation_id":"60c6ffac-10c6-440e-ac75-fe2fbdae59e4","resolution":{"observed_at":"2026-08-06T15:32:34.567976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T17:56:51.910988Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.15418","last_updated":"2025-08-21T10:20:00Z","snapshot_observed_at":"2026-08-08T09:52:34.305122Z","submitted_at":"2025-08-21T10:20:00Z","title":"LLaSO: A Foundational Framework for Reproducible Research in Large Language and Speech Model","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T17:56:51.910988Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2508.15418"},"observation_digest":"sha256:b8146cd6f21b9ed0f991328e8ac07344857bd314760f2c3eccb92edcb3eaf380","observation_id":"25c57aa8-a239-4e56-947f-112ad31e4aa3","resolution":{"observed_at":"2026-08-05T17:56:51.910988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T12:59:03.490856Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms.arXiv preprint arXiv:2411.15296, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.01106","last_updated":"2025-09-11T12:40:54Z","snapshot_observed_at":"2026-08-07T21:46:58.502468Z","submitted_at":"2025-09-01T03:53:47Z","title":"Robix: A Unified Model for Robot Interaction, Reasoning and Planning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T12:59:03.490856Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2509.01106"},"observation_digest":"sha256:6975189ce7749d641dba165da16e3e8bfb12f19206515ca92ecef3ddedc9d02b","observation_id":"aa0ca7aa-a81d-43a5-ba0b-89fa84c15f9e","resolution":{"observed_at":"2026-08-05T12:59:03.490856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2510.21828","last_updated":"2026-04-29T06:52:53Z","snapshot_observed_at":"2026-08-11T12:03:11.722209Z","submitted_at":"2025-10-22T02:23:40Z","title":"Structured and Abstractive Reasoning on Multi-modal Relational Knowledge Images","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-18T04:32:54.811976Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2510.21828"},"observation_digest":"sha256:17e11a0c5c5ff3a48dc1adb902626991415b12b79180381389d69c559e7b1761","observation_id":"591b2eb1-bb4e-48c7-8835-189caadd707a","resolution":{"observed_at":"2026-05-18T04:35:52.579252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-03T21:37:07.293897Z","title":"Mme-survey: A comprehensive sur- vey on evaluation of multimodal llms.arXiv preprint arXiv:2411.15296, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.14592","last_updated":"2026-07-18T14:25:41Z","snapshot_observed_at":"2026-08-07T09:15:50.580996Z","submitted_at":"2025-11-18T15:33:49Z","title":"DSBench: A Comprehensive Benchmark for Evaluating External and In-Cabin Risks","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T21:37:07.293897Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2511.14592"},"observation_digest":"sha256:b757d0b185e15774979a9d09201adc56e7cb4c9af61e0e9c9a03c6cb147f6270","observation_id":"44778109-6e37-4486-8e59-afb19081582d","resolution":{"observed_at":"2026-08-03T21:37:07.293897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2602.13294","last_updated":"2026-05-21T07:15:56Z","snapshot_observed_at":"2026-08-02T05:44:45.001749Z","submitted_at":"2026-02-09T05:46:44Z","title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T11:06:38.102348Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2602.13294"},"observation_digest":"sha256:e480371ddfdfafde656a3fb9d7444f29599570a2c6aa41da87dd5c5867816795","observation_id":"61c6187a-940a-49a1-a657-4ee3f4768b5d","resolution":{"observed_at":"2026-05-22T11:11:27.456373Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2603.11665","last_updated":"2026-04-21T11:59:18Z","snapshot_observed_at":"2026-08-03T00:46:37.430083Z","submitted_at":"2026-03-12T08:32:38Z","title":"Multi-Task Reinforcement Learning for Enhanced Multimodal LLM-as-a-Judge","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T12:34:53.596478Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2603.11665"},"observation_digest":"sha256:721aea76d9ce48e446028920bb3dd2680711ac2f0011328d4cb78d65f25d0800","observation_id":"6389e998-af92-4262-a4fe-d0ebddf7af20","resolution":{"observed_at":"2026-05-15T12:35:35.313567Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2603.11689","last_updated":"2026-07-01T08:11:59Z","snapshot_observed_at":"2026-08-09T19:37:08.421363Z","submitted_at":"2026-03-12T08:56:14Z","title":"Explicit Logic Channel for Validation and Enhancement of MLLMs on Zero-Shot Tasks","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-21T11:09:51.816554Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2603.11689"},"observation_digest":"sha256:4d93521c1097e268dbae71077c3362aa9dddc9470206b7952ad41a05591ff26c","observation_id":"166cd3b6-082f-4e88-a2df-0d5a66694532","resolution":{"observed_at":"2026-05-21T11:10:01.996695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2604.12424","last_updated":"2026-04-14T08:15:44Z","snapshot_observed_at":"2026-07-06T23:00:37.144068Z","submitted_at":"2026-04-14T08:15:44Z","title":"Decoding by Perturbation: Mitigating MLLM Hallucinations via Dynamic Textual Perturbation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T15:17:39.344348Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2604.12424"},"observation_digest":"sha256:a525f0459999799f21e86a9211dc75ad820570cb79f221a8dcf20c251d7f4869","observation_id":"d053b36e-308b-474a-a990-95989a370908","resolution":{"observed_at":"2026-05-11T10:51:03.774184Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2605.01359","last_updated":"2026-05-02T10:08:10Z","snapshot_observed_at":"2026-08-11T14:00:02.982088Z","submitted_at":"2026-05-02T10:08:10Z","title":"Structural Ranking of the Cognitive Plausibility of Computational Models of Analogy and Metaphors with the Minimal Cognitive Grid","version":1},"reference_index":196,"source":"arxiv_source","source_observed_at":"2026-05-09T14:33:11.033906Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2605.01359"},"observation_digest":"sha256:56f822f118fb47ddef80203e1371160d71e01f16bc69370430df6e9243c94003","observation_id":"808d609a-0137-4df7-8089-9000bd0006b0","resolution":{"observed_at":"2026-05-11T16:56:06.127142Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2605.03903","last_updated":"2026-05-05T15:56:12Z","snapshot_observed_at":"2026-08-10T21:36:00.852125Z","submitted_at":"2026-05-05T15:56:12Z","title":"CC-OCR V2: Benchmarking Large Multimodal Models for Literacy in Real-world Document Processing","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-07T16:18:35.484800Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2605.03903"},"observation_digest":"sha256:e56851296fd209387cd77a07be8c9da2ec65ccd1895133d1216bdc875acad64d","observation_id":"4f48086b-2a9e-480d-8aae-3b3209320359","resolution":{"observed_at":"2026-05-11T23:46:49.148170Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2605.11716","last_updated":"2026-05-12T08:05:10Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T08:05:10Z","title":"SafeSteer: A Decoding-level Defense Mechanism for Multimodal Large Language Models","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-13T06:56:10.053418Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2605.11716"},"observation_digest":"sha256:609dd717b6881b7a64e24b8cde58216ea586ca8e1b74dcb50a1c594d5335123f","observation_id":"87b8f2e0-f26f-423e-afcc-d44853e0151a","resolution":{"observed_at":"2026-05-13T06:57:27.658510Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2605.18018","last_updated":"2026-05-18T08:09:37Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:09:37Z","title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T12:10:54.874012Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2605.18018"},"observation_digest":"sha256:24c5a78182a88dbf98b5ceb1f8d8fea209c2243d5ebb47d15a82b35a810c6b43","observation_id":"c3cb4d90-09dd-4f00-87ec-90e7008fde86","resolution":{"observed_at":"2026-05-20T12:13:16.285669Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2605.18115","last_updated":"2026-05-18T09:24:39Z","snapshot_observed_at":"2026-08-06T04:43:08.432770Z","submitted_at":"2026-05-18T09:24:39Z","title":"WinTok: A Win-Win Hybrid Tokenizer via Decomposing Visual Understanding and Generation with Transferable Tokens","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T12:04:19.761430Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2605.18115"},"observation_digest":"sha256:749785210ec3336da421d0127ac529848e38fb2467ee457cd595ded42338cd9b","observation_id":"f47b0468-b02e-4423-90c4-a8e7b2ee9ee3","resolution":{"observed_at":"2026-05-20T12:08:15.747254Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2605.21479","last_updated":"2026-05-20T17:58:24Z","snapshot_observed_at":"2026-08-08T13:30:02.204396Z","submitted_at":"2026-05-20T17:58:24Z","title":"WikiVQABench: A Knowledge-Grounded Visual Question Answering Benchmark from Wikipedia and Wikidata","version":1},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-05-21T04:41:06.779529Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2605.21479"},"observation_digest":"sha256:7ab02245c7e132db043d9c8e0fa9dae54ff1d8ef42d0883db65e6be6ce57d71f","observation_id":"63ede929-1537-4a70-bc1f-a535b0984e76","resolution":{"observed_at":"2026-05-21T04:43:58.708522Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2606.27316","last_updated":"2026-06-25T17:29:58Z","snapshot_observed_at":"2026-07-31T08:21:54.248070Z","submitted_at":"2026-06-25T17:29:58Z","title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-26T03:56:28.271760Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2606.27316"},"observation_digest":"sha256:fad6c07460abb52ecf1294c62710564a72795f6889cee0632b7ccba239b20d70","observation_id":"1e5eda72-809b-41a2-97cb-c4920fb07921","resolution":{"observed_at":"2026-06-26T03:58:57.116684Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2411.15296","doi":"10.48550/arxiv.2411.15296","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms","venue":"arXiv (Cornell University)","work_id":"a5c11886-899f-4388-9772-29f96ab8de76","year":2024},"citing_paper":{"arxiv_id":"2606.27446","last_updated":"2026-06-25T18:17:04Z","snapshot_observed_at":"2026-08-11T07:00:39.586068Z","submitted_at":"2026-06-25T18:17:04Z","title":"Causal Connections: Leveraging Multilingual Fine-Tuning for Financial QA@FinCausal 2026","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-06-29T02:17:26.872854Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2606.27446"},"observation_digest":"sha256:cb03c8b97c796413f3c059c06158aa2835bd173f6777fbe24e49443084b28111","observation_id":"52278575-45e4-495c-8025-2304dd4f967b","resolution":{"observed_at":"2026-06-29T02:23:00.931773Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-07-14T12:45:30.225562Z","title":"arXiv preprint arXiv:2411.15296 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.10308","last_updated":"2026-07-11T13:28:16Z","snapshot_observed_at":"2026-08-06T08:09:32.911549Z","submitted_at":"2026-07-11T13:28:16Z","title":"Generalize LMMs to Versatile Visual Modalities via Fabricated Modality Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-14T12:45:30.225562Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2607.10308"},"observation_digest":"sha256:1d1245f58a761e5c2ee2bc2e9e85cb7abcd2e8ac8d5c5c551134e3425b3c272a","observation_id":"d3532870-03af-48ba-8a64-349c3088724e","resolution":{"observed_at":"2026-07-14T12:45:30.225562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-02T01:58:01.941030Z","title":"Mme-survey: A comprehensive survey on evaluation of multimodal llms.arXiv preprint arXiv:2411.15296, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14499","last_updated":"2026-07-16T02:25:03Z","snapshot_observed_at":"2026-08-07T14:02:26.886721Z","submitted_at":"2026-07-16T02:25:03Z","title":"Contextualized Evaluation of Vision Language Models through Dynamic, Multi-turn Interactions","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T01:58:01.941030Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2607.14499"},"observation_digest":"sha256:79ec01752c4fad32e239d85c005914b2c85cf66a66ae4bfe2ae40e2df9f8d084","observation_id":"5cbdc1b8-f069-4838-a1cb-7d9a07af4024","resolution":{"observed_at":"2026-08-02T01:58:01.941030Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15296","snapshot_observed_at":"2026-08-10T04:45:07.223493Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.07435","last_updated":"2026-08-07T17:21:04Z","snapshot_observed_at":"2026-08-11T13:13:36.409865Z","submitted_at":"2026-08-07T17:21:04Z","title":"SABRE: Scalable and Automated Benchmarking of VLMs under Stress","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-10T04:45:07.223493Z"},"links":{"cited_paper":"/paper/2411.15296","citing_paper":"/paper/2608.07435"},"observation_digest":"sha256:5b5368db6dd0c6fe791d4c33cf61f7e00a654df5477a64a04db48c0771c9c287","observation_id":"1a816af5-5c56-4df6-86ff-b56fda2106c0","resolution":{"observed_at":"2026-08-10T04:45:07.223493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2411.15296/citation-record","integrity":"/paper/2411.15296/integrity","json":"/paper/2411.15296/citation-record.json","paper":"/paper/2411.15296"},"outbound":[],"paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T09:14:56.498818Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 33 inbound Pith citation observations for arXiv:2411.15296."}