{"as_of":"2026-08-07T06:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2672c4bb25a4f8faa09be7dc420c7e8b482caedc1147914ef45f43ee3503212e","coverage":[{"denominator":238,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-20T06:20:36.235304Z","state":"measured"},{"denominator":156,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":156,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":56,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T06:02:30.485023Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T03:19:31.849614Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-05T10:34:24.268925Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-19T11:55:30.048525Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2406.08035"},"observation_digest":"sha256:6acdc4c87b8e1b4cdfc9d1fda2c8dee91658b2286d29f5686c7ae571af26c9b9","observation_id":"ab603d58-ef63-4ccd-abb8-47741ba195c2","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2411.16771","last_updated":"2026-04-23T13:21:19Z","snapshot_observed_at":"2026-08-02T03:20:43.405143Z","submitted_at":"2024-11-25T06:17:23Z","title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:12.821916Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2411.16771"},"observation_digest":"sha256:7650890d1dfec261c41dd488cb4dbd59354de196c39eb915891e54d54cb2a5b1","observation_id":"3608fb53-3b6e-426d-953a-e9f524d786fa","resolution":{"observed_at":"2026-05-23T16:58:12.099189Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-02T18:54:27.250149Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":143,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:057b828b268bdebd97bb2520436d77fe073a05e2855455c15cfbcee9c6efa246","observation_id":"a47649b8-0f06-49d9-97a9-8dddce700d69","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2501.12386","last_updated":"2025-07-13T18:57:17Z","snapshot_observed_at":"2026-08-06T07:17:05.291678Z","submitted_at":"2025-01-21T18:59:00Z","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T02:52:20.643070Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2501.12386"},"observation_digest":"sha256:315fd241ccd1d97bf1e84b1af8f965d592436c91437e2b604037b76847702afd","observation_id":"9710c9e3-ac62-487f-8521-303ae1e425ac","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2502.04326","last_updated":"2026-03-01T04:35:41Z","snapshot_observed_at":"2026-07-06T20:32:23.502480Z","submitted_at":"2025-02-06T18:59:40Z","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","version":3},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-17T05:53:26.066674Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2502.04326"},"observation_digest":"sha256:5362a056daf85398a654f8d90fd750faa85f5cb0734317f83ec171c6f40a639c","observation_id":"4a0c88b2-7a11-4f3c-9bc8-1d33ab1fbebb","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2505.14362","last_updated":"2026-03-01T04:59:56Z","snapshot_observed_at":"2026-08-02T12:23:34.946873Z","submitted_at":"2025-05-20T13:48:11Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-11T14:42:56.565621Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2505.14362"},"observation_digest":"sha256:21add015002f3d0a1034da4f46bc5cd5543cb8f795ddcf1dfe2d65696b4b2b94","observation_id":"3079ff5c-85cf-4863-9f30-621935021e31","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2505.16416","last_updated":"2026-05-21T10:32:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-22T09:05:01Z","title":"Circle-RoPE: Cone-like Decoupled Rotary Positional Embedding for Large Vision-Language Models","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-22T14:19:34.622854Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2505.16416"},"observation_digest":"sha256:e0bfcb978be3f51ac30d1771b1daf82b8276b6a3e78d009f26658344b22a9f4f","observation_id":"0668538a-cf7b-40e9-a8cd-75764d2ee628","resolution":{"observed_at":"2026-05-22T14:21:39.824392Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2506.05425","last_updated":"2026-04-28T02:01:09Z","snapshot_observed_at":"2026-07-31T07:37:26.215945Z","submitted_at":"2025-06-05T05:51:35Z","title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-19T11:36:36.687324Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2506.05425"},"observation_digest":"sha256:8a5b28b7af983ad9b915e90edafa233bddccd62eb38f7e0f089ad769ded8342a","observation_id":"a5f2bf78-fec8-4178-8f43-4d55feb0d08e","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2506.05831","last_updated":"2026-04-07T10:06:53Z","snapshot_observed_at":"2026-08-02T20:29:58.309302Z","submitted_at":"2025-06-06T07:56:41Z","title":"HeartcareGPT: A Unified Multimodal ECG Suite for Dual Signal-Image Modeling and Understanding","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-19T10:44:01.880405Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2506.05831"},"observation_digest":"sha256:e852b20cd113c16ac6075d9531f052ec49b0e138aff74bb826f7ab033a27299f","observation_id":"45b905aa-47e5-4ff6-b4e2-799dc5540401","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-07T06:02:30.485023Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06279","last_updated":"2025-06-06T17:59:06Z","snapshot_observed_at":"2026-08-07T05:54:42.681893Z","submitted_at":"2025-06-06T17:59:06Z","title":"CoMemo: LVLMs Need Image Context with Image Memory","version":1},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-08-07T06:02:30.485023Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2506.06279"},"observation_digest":"sha256:b00a76556bc0cda2734432ffccfa2e0ed0d20413f1640a9565987046db53ed78","observation_id":"87a056bb-2da8-4bad-b4e4-05219762e1c8","resolution":{"observed_at":"2026-08-07T06:02:30.485023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-07T00:25:01.561578Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14428","last_updated":"2025-06-17T11:45:33Z","snapshot_observed_at":"2026-08-07T00:15:32.812600Z","submitted_at":"2025-06-17T11:45:33Z","title":"Toward Rich Video Human-Motion2D Generation","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T00:25:01.561578Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2506.14428"},"observation_digest":"sha256:08a8d73846f9a64999c72e9163537ca4ef2469d87133b22de60741aff970cfdb","observation_id":"559a4317-2615-4f68-9f49-785dc8100937","resolution":{"observed_at":"2026-08-07T00:25:01.561578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-06T23:11:54.884075Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.19513","last_updated":"2025-06-24T11:03:10Z","snapshot_observed_at":"2026-08-06T23:04:27.616711Z","submitted_at":"2025-06-24T11:03:10Z","title":"Visual hallucination detection in large vision-language models via evidential conflict","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T23:11:54.884075Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2506.19513"},"observation_digest":"sha256:b0c49c724d2db99b3f1b585b35caeb5a0527e6c836a87ab7c5665f579d5ffbfd","observation_id":"ed19ce15-88c0-4f53-b8aa-b72b8f086db0","resolution":{"observed_at":"2026-08-06T23:11:54.884075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2507.00748","last_updated":"2026-04-12T11:20:16Z","snapshot_observed_at":"2026-07-31T10:24:51.553367Z","submitted_at":"2025-07-01T13:48:57Z","title":"Improving the Reasoning of Multi-Image Grounding in MLLMs via Reinforcement Learning","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-19T06:50:02.607136Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2507.00748"},"observation_digest":"sha256:6fff668398f353a3d3ea56ab7c7a63f90bafd3e22c5e337d14142467bef017f8","observation_id":"f970a53b-403c-422c-932f-5ba58f3ec871","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-06T19:25:02.971505Z","title":"mplug-owl3: Towards long image-sequence understanding 11 in multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06272","last_updated":"2025-08-09T05:40:33Z","snapshot_observed_at":"2026-08-06T19:16:35.536045Z","submitted_at":"2025-07-08T07:46:26Z","title":"LIRA: Inferring Segmentation in Large Multi-modal Models with Local Interleaved Region Assistance","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:02.971505Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2507.06272"},"observation_digest":"sha256:435574cdcf0212f7d14db0ad1accde3443aecefb24cba80757b1af1f7a598f66","observation_id":"f156745b-ab03-44ca-becf-967ee861c55b","resolution":{"observed_at":"2026-08-06T19:25:02.971505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-06T12:12:23.706791Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.22003","last_updated":"2025-07-30T04:41:52Z","snapshot_observed_at":"2026-08-06T12:12:22.990814Z","submitted_at":"2025-07-29T16:53:27Z","title":"See Different, Think Better: Visual Variations Mitigating Hallucinations in LVLMs","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T12:12:23.706791Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2507.22003"},"observation_digest":"sha256:cb7d128c810d612e68433e01217120e99264574608f4f8fc86bf432ce6c590a8","observation_id":"b3d8d813-aa58-4a52-9185-6336b493b57c","resolution":{"observed_at":"2026-08-06T12:12:23.706791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-06T05:51:16.088023Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01236","last_updated":"2025-08-02T07:22:08Z","snapshot_observed_at":"2026-08-06T05:51:15.223687Z","submitted_at":"2025-08-02T07:22:08Z","title":"Mitigating Information Loss under High Pruning Rates for Efficient Large Vision Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T05:51:16.088023Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2508.01236"},"observation_digest":"sha256:03c5a45c9bd74e1c068babc74bae6019e3ebc2ecfed1338e8e72c602185ce1f7","observation_id":"053c4317-01a0-4b80-b2ee-b2bdaf1627d4","resolution":{"observed_at":"2026-08-06T05:51:16.088023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-06T04:49:42.097293Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03039","last_updated":"2025-08-05T03:33:24Z","snapshot_observed_at":"2026-08-06T04:49:36.808571Z","submitted_at":"2025-08-05T03:33:24Z","title":"VideoForest: Person-Anchored Hierarchical Reasoning for Cross-Video Question Answering","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T04:49:42.097293Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2508.03039"},"observation_digest":"sha256:079c751bee2aa21b20e5d072bec248907b30dfdd3939d9f55d747cab5c925c50","observation_id":"7633f68b-5142-46fb-be42-0dcf7ba2cfcf","resolution":{"observed_at":"2026-08-06T04:49:42.097293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-05T22:35:50.952229Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06895","last_updated":"2025-08-09T09:00:45Z","snapshot_observed_at":"2026-08-05T22:34:59.302688Z","submitted_at":"2025-08-09T09:00:45Z","title":"BASIC: Boosting Visual Alignment with Intrinsic Refined Embeddings in Multimodal Large Language Models","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-05T22:35:50.952229Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2508.06895"},"observation_digest":"sha256:4de1329f390810e8ea07099f0283793e72af20937e9f5879d3b24042253241f9","observation_id":"1c631496-3b3d-46f0-b4de-2b6e88a6785a","resolution":{"observed_at":"2026-08-05T22:35:50.952229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-05T23:32:17.804125Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10922","last_updated":"2025-08-07T08:52:11Z","snapshot_observed_at":"2026-08-06T10:35:41.875223Z","submitted_at":"2025-08-07T08:52:11Z","title":"A Survey on Video Temporal Grounding with Multimodal Large Language Model","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-05T23:32:17.804125Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2508.10922"},"observation_digest":"sha256:f2c388bc3258fa89d31b3d1c8e04d2550d999f8c8e4ecf5f540de66ba0e77e81","observation_id":"1949ae06-50bd-4c89-a926-48883de00917","resolution":{"observed_at":"2026-08-05T23:32:17.804125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-05T10:15:50.299374Z","title":"mPLUG- Owl3: Towards Long Image-Sequence Understanding in Multi-Modal LargeLanguageModels,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04378","last_updated":"2025-09-09T08:09:40Z","snapshot_observed_at":"2026-08-05T10:15:47.702669Z","submitted_at":"2025-09-04T16:40:15Z","title":"Aesthetic Image Captioning with Saliency Enhanced MLLMs","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T10:15:50.299374Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2509.04378"},"observation_digest":"sha256:4964a7229672b681d97bbe78f629f4543f29e1044ae543a29e4b9492f213f448","observation_id":"0913bc48-a0bf-460e-a289-d19558f33f32","resolution":{"observed_at":"2026-08-05T10:15:50.299374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-04T21:27:36.085495Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07680","last_updated":"2025-09-09T17:59:39Z","snapshot_observed_at":"2026-08-04T21:27:30.632810Z","submitted_at":"2025-09-09T17:59:39Z","title":"CAViAR: Critic-Augmented Video Agentic Reasoning","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-04T21:27:36.085495Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2509.07680"},"observation_digest":"sha256:4630fcfabb5a9520e55386beaca8aeb8f071b907aceb7d6dad52b223f9e1db7e","observation_id":"ee04e35f-2e5d-4c03-a0ba-e946184a1057","resolution":{"observed_at":"2026-08-04T21:27:36.085495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-04T19:27:40.837413Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09254","last_updated":"2025-09-11T08:39:08Z","snapshot_observed_at":"2026-08-04T19:27:38.196956Z","submitted_at":"2025-09-11T08:39:08Z","title":"Towards Better Dental AI: A Multimodal Benchmark and Instruction Dataset for Panoramic X-ray Analysis","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-04T19:27:40.837413Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2509.09254"},"observation_digest":"sha256:9c54d70478f0d4f567ce6003341cde3a08ba7f98c068683125d056948dad207d","observation_id":"58f36d98-be96-4120-a13d-9d454a7a221c","resolution":{"observed_at":"2026-08-04T19:27:40.837413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2509.15602","last_updated":"2026-04-16T06:31:32Z","snapshot_observed_at":"2026-08-03T02:26:40.337656Z","submitted_at":"2025-09-19T05:08:05Z","title":"TennisTV: Do Multimodal Large Language Models Understand Tennis Rallies?","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-18T16:40:16.630602Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2509.15602"},"observation_digest":"sha256:50853a452c9aadf59f59dae53663d8080aa54d4eb256dedc56cc84bff8f7c2ce","observation_id":"5fdfa497-be9c-43d1-956c-9b4e82d77196","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-04T11:32:15.178796Z","title":"Preprint, arXiv:2408.04840","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.04514","last_updated":"2026-06-08T20:02:58Z","snapshot_observed_at":"2026-08-04T11:32:07.464063Z","submitted_at":"2025-10-06T06:05:36Z","title":"ChartAgent: A Multimodal Agent for Visually Grounded Reasoning in Complex Chart Question Answering","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T11:32:15.178796Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2510.04514"},"observation_digest":"sha256:693afcea5f6bdf3b2c24d5eca3a6722f939669bf3b8bcf19560476606fdd50b7","observation_id":"a095ebf5-ebae-4d08-af40-75a0eb04acbe","resolution":{"observed_at":"2026-08-04T11:32:15.178796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2510.20093","last_updated":"2026-04-14T09:32:02Z","snapshot_observed_at":"2026-07-06T22:33:53.834577Z","submitted_at":"2025-10-23T00:27:32Z","title":"StableSketcher: Enhancing Diffusion Model for Pixel-based Sketch Generation via Visual Question Answering Feedback","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-18T05:26:42.034972Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2510.20093"},"observation_digest":"sha256:5c50c2801febff81513cefe2ad02edf0ab5c4da283b6ca8ed138e6f5399b96ae","observation_id":"6ae7cd08-9b32-4089-b065-796ae06b34bd","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-03T20:23:08.154037Z","title":"12 mplug-owl3: Towards long image-sequence understand- ing in multi-modal large language models.arXiv preprint arXiv:2408.04840, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.20272","last_updated":"2026-07-03T06:27:17Z","snapshot_observed_at":"2026-08-03T20:22:57.328566Z","submitted_at":"2025-11-25T12:58:32Z","title":"VKnowU: Evaluating Visual Knowledge Understanding in Multimodal LLMs","version":2},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-03T20:23:08.154037Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2511.20272"},"observation_digest":"sha256:256f0ae4093a5ef1fe6800e4042508020e61ceccf11918a1694d0dde199f0fee","observation_id":"501df437-b224-4a3c-ab3a-14689e3b59b8","resolution":{"observed_at":"2026-08-03T20:23:08.154037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-02T23:33:15.573793Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models.arXiv preprint arXiv:2408.04840,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.13602","last_updated":"2026-05-30T00:19:02Z","snapshot_observed_at":"2026-08-02T23:33:07.026705Z","submitted_at":"2026-02-14T04:52:11Z","title":"Towards Sparse Video Understanding and Reasoning","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-02T23:33:15.573793Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2602.13602"},"observation_digest":"sha256:04b74cfc1c34ac9280e263697b8abd70ea8301f3067c451bbe03cca3fc03c536","observation_id":"e3a92804-3732-4cd5-8eae-fb0ad171bbf2","resolution":{"observed_at":"2026-08-02T23:33:15.573793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-02T20:19:17.457027Z","title":"arXiv preprint arXiv:2408.04840 (2024) 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.23615","last_updated":"2026-07-08T09:59:15Z","snapshot_observed_at":"2026-08-04T03:41:58.938743Z","submitted_at":"2026-02-27T02:43:35Z","title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-02T20:19:17.457027Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2602.23615"},"observation_digest":"sha256:413f33fe554f6b98cdac8e114263d5b432172a4753f711634477a3937ee92e40","observation_id":"f85148e9-91de-480b-902b-788e24b75114","resolution":{"observed_at":"2026-08-02T20:19:17.457027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:dd86eb7a31343473f7c0a86aa1af21f02f3d37fd21421eb247c7c41c3baae81a","observation_id":"8bdc85bc-4d4f-4bed-bba9-55ff53be74ec","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2604.04379","last_updated":"2026-04-06T03:01:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-06T03:01:52Z","title":"Reinforce to Learn, Elect to Reason: A Dual Paradigm for Video Reasoning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T19:40:41.642852Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2604.04379"},"observation_digest":"sha256:4c646aad6f53e79a04bd1872d1062c3e2120e7ca8a05f364be35a943fa686bd4","observation_id":"d6fdfb52-2834-45a6-a785-66b9afb3be91","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2604.05418","last_updated":"2026-04-16T09:51:53Z","snapshot_observed_at":"2026-08-03T01:38:19.340462Z","submitted_at":"2026-04-07T04:26:59Z","title":"VideoStir: Understanding Long Videos via Spatio-Temporally Structured and Intent-Aware RAG","version":3},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-10T19:15:02.124035Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2604.05418"},"observation_digest":"sha256:5ab0f3a69ff0b61b40d99c60ccf0a8797f558584d403dc95433691620542a3b3","observation_id":"ffa1a70f-95ba-49aa-b788-acc41fcdbb5d","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2604.07914","last_updated":"2026-04-09T07:31:27Z","snapshot_observed_at":"2026-07-06T22:57:09.435331Z","submitted_at":"2026-04-09T07:31:27Z","title":"Mitigating Entangled Steering in Large Vision-Language Models for Hallucination Reduction","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T18:16:31.701345Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2604.07914"},"observation_digest":"sha256:452a10a77578675037a7e2a0b56904354bb2f7f9709bb2e4f51bdd8d10dd3afc","observation_id":"23f62cc7-2ed3-42eb-a396-fdceca16c285","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2604.14149","last_updated":"2026-04-16T15:48:38Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:59:52Z","title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-10T13:28:58.920442Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2604.14149"},"observation_digest":"sha256:779b61f5d324d3f1f73779c214dddfb63f33167684e496e2bcfa687bed9f55bd","observation_id":"927cbe4f-7be5-4c77-b890-50ba9565aaff","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2604.17087","last_updated":"2026-04-18T17:52:02Z","snapshot_observed_at":"2026-07-06T23:04:14.688737Z","submitted_at":"2026-04-18T17:52:02Z","title":"EvoComp: Learning Visual Token Compression for Multimodal Large Language Models via Semantic-Guided Evolutionary Labeling","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T07:00:36.870817Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2604.17087"},"observation_digest":"sha256:902af28071ec067d60747dbbd0c12e7e51ff64c7722c4ff5e7952330989db0a2","observation_id":"8f4f5e7d-70bf-4450-9772-9bf3754028aa","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2604.22498","last_updated":"2026-04-24T12:26:49Z","snapshot_observed_at":"2026-07-06T23:08:51.913995Z","submitted_at":"2026-04-24T12:26:49Z","title":"CGC: Compositional Grounded Contrast for Fine-Grained Multi-Image Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-08T12:26:01.568507Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2604.22498"},"observation_digest":"sha256:0b08467f69a472e7aee70ee28c07956b387aae5f9122f92d558f60eb542401ca","observation_id":"9597da25-df58-4660-b2b9-c90cdd08763e","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.05848","last_updated":"2026-05-08T13:46:44Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T08:23:27Z","title":"VideoRouter: Query-Adaptive Dual Routing for Efficient Long-Video Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-08T14:48:39.444933Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.05848"},"observation_digest":"sha256:3c8ed6824201ddd54e9da45508d98f2d6112968fdbf3e245eeb1d82582326a9b","observation_id":"1bbecdb9-12e7-4b4c-b4b5-05754e2e008e","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.05848","last_updated":"2026-05-08T13:46:44Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T08:23:27Z","title":"VideoRouter: Query-Adaptive Dual Routing for Efficient Long-Video Understanding","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-11T01:57:42.822121Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.05848"},"observation_digest":"sha256:63fc35ffefbceaddc27567ca53ec9facd26a212553677074f3a8447241093f73","observation_id":"9148bc60-ff90-4f24-81f2-c5265907d705","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.18160","last_updated":"2026-06-02T05:15:38Z","snapshot_observed_at":"2026-07-06T23:29:04.241654Z","submitted_at":"2026-05-18T10:04:22Z","title":"Vision Inference Former: Sustaining Visual Consistency in Multimodal Large Language Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-20T11:38:01.947806Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.18160"},"observation_digest":"sha256:784d458649ed479d961334aec44ecc1204bb7043d32d1f87e882e598efb14d36","observation_id":"7792c837-fc3c-4a1f-9ede-36a6bdadac0f","resolution":{"observed_at":"2026-05-20T11:38:14.420098Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.18160","last_updated":"2026-06-02T05:15:38Z","snapshot_observed_at":"2026-07-06T23:29:04.241654Z","submitted_at":"2026-05-18T10:04:22Z","title":"Vision Inference Former: Sustaining Visual Consistency in Multimodal Large Language Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-30T18:42:46.422756Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.18160"},"observation_digest":"sha256:b28fc6ae73d6144fcfd0b56d20609a4387b1accb6e80716e8e102ae1ae61ab50","observation_id":"2ffc2cab-dc13-4d21-937f-9643240ad600","resolution":{"observed_at":"2026-06-30T18:45:00.167951Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.19846","last_updated":"2026-05-23T07:31:13Z","snapshot_observed_at":"2026-08-02T22:01:11.866294Z","submitted_at":"2026-05-19T13:40:26Z","title":"FineBench: Benchmarking and Enhancing Vision-Language Models for Fine-grained Human Activity Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T06:08:12.887394Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.19846"},"observation_digest":"sha256:0adeedfe09a80e750873c2e27d5411bbce368aba09caf6a3b02dc36fafbf9c49","observation_id":"403222ab-1573-4d5c-81bc-2845bbe8592c","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.19846","last_updated":"2026-05-23T07:31:13Z","snapshot_observed_at":"2026-08-02T22:01:11.866294Z","submitted_at":"2026-05-19T13:40:26Z","title":"FineBench: Benchmarking and Enhancing Vision-Language Models for Fine-grained Human Activity Understanding","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T07:59:57.488107Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.19846"},"observation_digest":"sha256:88c8672dfbef5ce60a293efc4a5f809f92b0265a54a6b8e35f03904186c49b1b","observation_id":"c2c5ce59-729d-4f4a-8236-b15ecf5596c9","resolution":{"observed_at":"2026-05-21T08:04:02.793257Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.19846","last_updated":"2026-05-23T07:31:13Z","snapshot_observed_at":"2026-08-02T22:01:11.866294Z","submitted_at":"2026-05-19T13:40:26Z","title":"FineBench: Benchmarking and Enhancing Vision-Language Models for Fine-grained Human Activity Understanding","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-30T18:09:49.178090Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.19846"},"observation_digest":"sha256:3cf6a86b3e8d3fb28f45ce39b7c480ae973fcad5e1551706e881decae66cf918","observation_id":"951f65a9-c13b-4aa4-986f-4bcf0e2b1742","resolution":{"observed_at":"2026-06-30T18:15:00.001853Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.20165","last_updated":"2026-05-19T17:50:25Z","snapshot_observed_at":"2026-07-06T23:30:49.211483Z","submitted_at":"2026-05-19T17:50:25Z","title":"CaMo: Camera Motion Grounded Evaluation and Training for Vision-Language Models","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-20T05:27:30.938311Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.20165"},"observation_digest":"sha256:231a61d36f78470d14dfa8fae34e8f6b3785b94f2593864fcd616b21ed247507","observation_id":"87c95789-6104-4b55-820f-e537c2d8b029","resolution":{"observed_at":"2026-05-20T06:20:36.894768Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2605.27959","last_updated":"2026-05-28T03:13:45Z","snapshot_observed_at":"2026-07-06T23:37:36.244456Z","submitted_at":"2026-05-27T04:52:42Z","title":"ROVER: Routing Object-Centric Visual Evidence for Grounded Multi-Image Reasoning","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-29T13:41:44.049230Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2605.27959"},"observation_digest":"sha256:dd66b648e49b4a929f569e266ef836ad9e28671c5f9981a6d581dc13a7a92d12","observation_id":"964e3731-771f-435f-8a57-93418b89f1c2","resolution":{"observed_at":"2026-06-29T13:43:28.638957Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2606.04591","last_updated":"2026-06-03T08:29:43Z","snapshot_observed_at":"2026-07-06T23:44:42.700120Z","submitted_at":"2026-06-03T08:29:43Z","title":"Fine-grained Fragment Retrieval in Multi-modal Long-form Dialogues","version":1},"reference_index":139,"source":"arxiv_source","source_observed_at":"2026-06-28T06:04:28.939248Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2606.04591"},"observation_digest":"sha256:02b71ca535507b403d11aaead88db088151fb0ba16522f0150fe39bf70f21be3","observation_id":"63407566-6e03-4694-b112-aadf20a542e6","resolution":{"observed_at":"2026-07-02T08:26:48.004528Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2606.12125","last_updated":"2026-06-10T14:19:15Z","snapshot_observed_at":"2026-08-04T05:43:14.707313Z","submitted_at":"2026-06-10T14:19:15Z","title":"Q-Fold: Query-Aware Focus-Context Spatio-Temporal Folding for Long Video Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T10:04:29.739632Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2606.12125"},"observation_digest":"sha256:7128e7020bab00be57ac3c2c94036c009729487a13f9e6d86f1ed73b2f80a47c","observation_id":"31aa1e9b-fa69-43f1-90f2-2efef04227e7","resolution":{"observed_at":"2026-07-03T10:27:56.051074Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":193,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:ffc389f34f7d3adf8458965831bd2253a0a401af3ceea80264326e8139222ace","observation_id":"894ca6ba-4f74-45fa-8363-550f21fe0e9f","resolution":{"observed_at":"2026-07-03T10:48:02.993392Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2606.17030","last_updated":"2026-06-17T13:54:57Z","snapshot_observed_at":"2026-08-01T21:48:31.832288Z","submitted_at":"2026-06-15T17:52:31Z","title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","version":3},"reference_index":169,"source":"arxiv_source","source_observed_at":"2026-06-27T04:19:26.332718Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2606.17030"},"observation_digest":"sha256:e345733799a710f10918be9f5397b64d7098cd0414bec4c6db14198941c75f60","observation_id":"b079ded5-78d6-4ce9-ad7c-f9b5973ba6f1","resolution":{"observed_at":"2026-07-03T17:18:43.794018Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2606.20077","last_updated":"2026-06-18T10:52:49Z","snapshot_observed_at":"2026-08-04T07:49:21.370818Z","submitted_at":"2026-06-18T10:52:49Z","title":"The Hidden Evolution of Disguised Visual Context inside the VLM","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-06-26T18:08:56.044278Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2606.20077"},"observation_digest":"sha256:f16db03d615507197f935038d07971d2f0d7a615f029f30a5f94c5cb670e5b1f","observation_id":"890decec-bec6-430e-9fd1-c85b2dbc2cca","resolution":{"observed_at":"2026-07-04T03:19:31.851640Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2606.28401","last_updated":"2026-06-24T11:06:22Z","snapshot_observed_at":"2026-07-07T00:02:29.000757Z","submitted_at":"2026-06-24T11:06:22Z","title":"Vision-driven Preference Synthesis for Mitigating Hallucinations in VLMs","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-30T01:22:16.176398Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2606.28401"},"observation_digest":"sha256:4920f08fdca5c46aee48f1171d9d8e77e848b28c7645a3520e28b2b47345cdb9","observation_id":"f301847e-620a-4a7f-87ba-f8462e2c1386","resolution":{"observed_at":"2026-07-01T15:35:47.610810Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":"2408.04840","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-04T03:19:31.849614Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","venue":"cs.CV","work_id":"6fde0828-1494-4a4e-8aec-3747de70602e","year":2024},"citing_paper":{"arxiv_id":"2607.00983","last_updated":"2026-07-01T14:19:24Z","snapshot_observed_at":"2026-07-07T00:06:35.457630Z","submitted_at":"2026-07-01T14:19:24Z","title":"QCA: Query- and Content-Aware Keyframe Selection for Long Video Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-02T14:09:54.549499Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2607.00983"},"observation_digest":"sha256:aa366c59fe0c082b9dda1fb16f674551e5ddb1c0acdcde4ac60bcd1896c2d42f","observation_id":"36ae4100-0c8d-4190-a485-ebe27b5e2fd8","resolution":{"observed_at":"2026-07-02T14:17:02.655589Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-11T12:10:52.100410Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04872","last_updated":"2026-07-06T09:47:23Z","snapshot_observed_at":"2026-08-05T15:57:57.173805Z","submitted_at":"2026-07-06T09:47:23Z","title":"EventCoT: Event-centric Video Chain-of-thought for Reasoning Temporal Localization","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-07-11T12:10:52.100410Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2607.04872"},"observation_digest":"sha256:b4135ed85a29d35bb0e2ff433125d2f2c93fdcfad14c37aa7d4e5b211eed5303","observation_id":"e2bfbc0b-1aa7-4e6e-ad9d-1eb2e65a6d43","resolution":{"observed_at":"2026-07-11T12:10:52.100410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-07-14T03:31:19.309532Z","title":"arXiv preprint arXiv:2408.04840 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11738","last_updated":"2026-07-13T16:00:03Z","snapshot_observed_at":"2026-08-06T12:36:13.363835Z","submitted_at":"2026-07-13T16:00:03Z","title":"Qwen-Audio-VAE Technical Report","version":1},"reference_index":182,"source":"arxiv_source","source_observed_at":"2026-07-14T03:31:19.309532Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2607.11738"},"observation_digest":"sha256:fa8917a541dc4241ad549be2c177057e28d89d01f72a5a8c3d0860433970fbb6","observation_id":"5bec75ba-1f6f-470b-b336-41f21e2fad57","resolution":{"observed_at":"2026-07-14T03:31:19.309532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-01T22:28:27.815322Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15752","last_updated":"2026-07-17T08:40:18Z","snapshot_observed_at":"2026-08-01T22:28:21.209330Z","submitted_at":"2026-07-17T08:40:18Z","title":"Personalized Image Aesthetic Assessment via Preference-rich Sample Mining and Cohort Merging","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-01T22:28:27.815322Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2607.15752"},"observation_digest":"sha256:c7f6496e6710919b381a0b27c4c9c1e0aae284db99130cc920b07a0f2c975045","observation_id":"4b1d9cc7-3b82-4c1c-8637-bccc5f2dcb28","resolution":{"observed_at":"2026-08-01T22:28:27.815322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-01T08:37:41.845892Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21061","last_updated":"2026-07-23T08:52:11Z","snapshot_observed_at":"2026-08-01T08:37:35.605662Z","submitted_at":"2026-07-23T08:52:11Z","title":"MVEI & EmObserver: Empowering MLLM-Oriented Visual Emotional Intelligence via Emotion Statement Judgement","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-01T08:37:41.845892Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2607.21061"},"observation_digest":"sha256:dfc9aa9220d199c8cd3671791d36428033294cef8a1f07bb830c89bdcfd5592c","observation_id":"23b177a1-d927-4d62-913f-afada9d78050","resolution":{"observed_at":"2026-08-01T08:37:41.845892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04840","snapshot_observed_at":"2026-08-05T21:53:29.121248Z","title":"mplug-owl3: Towards long image-sequence understanding in multi-modal large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.03279","last_updated":"2026-08-04T07:53:14Z","snapshot_observed_at":"2026-08-07T06:16:08.673788Z","submitted_at":"2026-08-04T07:53:14Z","title":"3DGSI-Assessor: A Large-Scale Dataset and An LMM-based Method for 3D Gaussian Splatting Image Quality Assessment","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T21:53:29.121248Z"},"links":{"cited_paper":"/paper/2408.04840","citing_paper":"/paper/2608.03279"},"observation_digest":"sha256:3a6f16937c2148c90a0e8b6158a9c3cd4fbb9b665db58b6661e2b7e95fe66336","observation_id":"13d78038-614a-4f76-b6ee-4e875293a278","resolution":{"observed_at":"2026-08-05T21:53:29.121248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.04840/citation-record","integrity":"/paper/2408.04840/integrity","json":"/paper/2408.04840/citation-record.json","paper":"/paper/2408.04840"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:07:05.804778Z","title":"Scaling Learning Algorithms Towards","venue":null,"work_id":"bb2761cc-98d0-411b-92f6-803773d64460","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:affd8341f06a366e9dc4506ae6ba83caefb9eea26cf9054aa11dd653fb94358d","observation_id":"d152c559-f91b-46f6-83ba-2551ba297dd3","resolution":{"observed_at":"2026-05-20T06:20:36.700677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:07:05.766846Z","title":"and Osindero, Simon and Teh, Yee Whye , journal =","venue":null,"work_id":"0a5921e3-ac4e-46f1-85ae-866119a87be0","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:a9d6afb48661dabb27adad3ee87bd075bc36b1943fdb56335db71f2040b59050","observation_id":"00ec488e-6ab0-4a15-a07a-7098679feaf7","resolution":{"observed_at":"2026-05-20T06:20:36.798740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T21:57:45.913036Z","title":"2016 , publisher=","venue":null,"work_id":"cf0899e0-53ee-4591-aae4-f38fa5ac12ad","year":2016},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:81a44c047dfdb5eb887ed143448d10b1627599425586ac1751ca8d93a429de95","observation_id":"a2ad1236-f8a6-488b-b6f1-10d7dad77009","resolution":{"observed_at":"2026-05-20T06:20:36.801465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision , pages=","venue":null,"work_id":"9ce5ad10-5e4d-4c63-8e19-511452ddb851","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:c279276e0cc1246e9670e0f893f6e4840deacb751efd56144f62f5207423482d","observation_id":"88700bb6-3084-4318-9425-781e7a465ddc","resolution":{"observed_at":"2026-05-20T06:20:36.804054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":"arXiv (Cornell University)","work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:6a078744c4f1216735981ab67f2bb04cd8cc66c92472ad1558dc169e0ded95ea","observation_id":"1c2852a6-e767-40dd-90b9-a368384dc822","resolution":{"observed_at":"2026-05-20T06:20:36.308851Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"94e9b425-d12d-489a-b55f-ca55e906e332","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:9ce59138d5fbcc5e9338312a215fe81dfd721762ee00b62d0c34f65655857ac8","observation_id":"94ade3e2-91ae-4a81-8725-7d45ecf2d541","resolution":{"observed_at":"2026-05-20T06:20:36.806512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"b4c0b284-a364-48c4-8e5d-93d8e3bb7e0d","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:991bfd65b36cdd7cc26dcd32c3e42a05523a1efd377816177dcd60a13e5ac585","observation_id":"46833eb5-afb4-4337-8c04-6031360c0806","resolution":{"observed_at":"2026-05-20T06:20:36.810565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2023 , url=","venue":null,"work_id":"7fb942b8-285b-463c-a855-af0dec2e6040","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:427ac04048e26ce79a03c7fda67f12dbb09ae5ae2fe95bc4a1185649d6396699","observation_id":"3a70f8a3-f81a-487c-99e5-9bc5df9289bb","resolution":{"observed_at":"2026-05-20T06:20:36.813234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T05:40:44.344100Z","title":"ArXiv , year=","venue":null,"work_id":"8a8b63b4-c22e-413d-88f5-8753fc5f8402","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:0906ad55b42d69ba11570c157dd5a84ec63f85c77aee6914ea889b7b484ed844","observation_id":"753a7c4b-2dbe-4119-9bfc-4f2e1d7a3ca5","resolution":{"observed_at":"2026-05-20T06:20:36.816796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T11:06:11.425440Z","title":"ArXiv , year=","venue":null,"work_id":"abef9ec9-cc35-48c6-968c-28f788c4162f","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:7b71b42055ee4fe061849ce35861b2ce7f09c07e91ae8c1b5e5a6bfb4e6fcd64","observation_id":"f14e61f1-7c5a-40c1-a6e1-cc2918ad9c63","resolution":{"observed_at":"2026-05-20T06:20:36.818910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"cc6a934c-0dfc-4b87-b4c9-f1a1b7e4c5c4","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:e5c47e039102629167d5a68a29ab3433bea94464846758b05e6bec5785648b26","observation_id":"3e935220-6d7a-4359-9129-931142f84d12","resolution":{"observed_at":"2026-05-20T06:20:36.820819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:22:59.197465Z","title":"ArXiv , year=","venue":null,"work_id":"0b0a68e3-2403-4c55-9345-8b2ec2e091e3","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:1585269b14d865cd459407f9a0bf2b9196ff4685494302359eedd19f29e0fc1b","observation_id":"2d26736a-cca4-4234-9531-4ece34d31204","resolution":{"observed_at":"2026-05-20T06:20:36.822855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":"2d02a6b4-0738-4824-a209-feca5d3e1251","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:1ff00ee187997b51b5f6e41d2c2c4e780db5a1d9ba03da05b0261633377f73f8","observation_id":"421a475f-44f5-4c04-ae7e-dbac6f0d756e","resolution":{"observed_at":"2026-05-20T06:20:36.825014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"b8754feb-4852-49ee-9366-0ecc2c05c558","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:ccd5fb7a3b2a236781c2649ec075a068fcbb76f1940082c7e4ff7052148a4839","observation_id":"ec1cfc84-7bf1-473b-b461-2c28a5ba1a12","resolution":{"observed_at":"2026-05-20T06:20:36.827390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:22:59.077174Z","title":"ArXiv , year=","venue":null,"work_id":"6a0a5e07-6650-4ae5-8df0-34433c40ea24","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:281837407832a10aa4fb223a8555e75fb9e7295fe61a93227017e0ad0a78b134","observation_id":"3183f8bd-6770-4ead-b564-3a2a86d224ed","resolution":{"observed_at":"2026-05-20T06:20:36.829394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:22:59.073628Z","title":"ArXiv , year=","venue":null,"work_id":"ed161624-e3e5-4a4d-abf7-b6b89e4cc485","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:93c7ab1ad876c590a7326104ddeb03cf0c42dd8bb25ad5ecab02ecda4afe3e3b","observation_id":"feae5519-5a67-47ea-9151-94055b25e7b9","resolution":{"observed_at":"2026-05-20T06:20:36.831381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":"e4b1e266-53b0-4b74-82b5-6fea14684e0d","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:5f23c5685796c2bf9a18bde5914c664457951b6c9d60aba33d628b4adbe77a0c","observation_id":"c639c225-477e-42a1-af18-42444ea4011a","resolution":{"observed_at":"2026-05-20T06:20:36.833781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"a23b6d7d-8926-4cf7-ba92-b89f842fc4b6","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:729584fb1a3104076de49767c1200a35b3dcc93fa95bca7199233a7f41610fb5","observation_id":"8e9277c1-f2b2-412f-bcd0-450571db335c","resolution":{"observed_at":"2026-05-20T06:20:36.835831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"17e8fab6-9fb5-41b9-b9ee-91e049a0adc0","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:f1275e6f79c49d957f3d55479f07d30b1ef44bebce415c20f7a8cffcfdaf1435","observation_id":"2239683f-121c-44ed-bc35-8fe1192c9202","resolution":{"observed_at":"2026-05-20T06:20:36.837718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:22:59.086057Z","title":"International Conference on Machine Learning , year=","venue":null,"work_id":"19429023-f7d8-41b2-a024-5fc6ec74e4b6","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:6cc62420998e68483d09f73c349bae3be90fe5a24fbac0a0d59a1b844f75077c","observation_id":"e3d8ce9d-ee23-481c-b8ab-4e454a72f78d","resolution":{"observed_at":"2026-05-20T06:20:36.839697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":"248ee7ca-adfc-4f87-b2d2-69a277dbb2f5","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:30b37f568ec7574b430a05d8baeae72ebecad7d70e91a01a26b339e194f8e51d","observation_id":"c3ee6392-4914-41c0-a32c-5793cdbd2768","resolution":{"observed_at":"2026-05-20T06:20:36.841798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"cited_work":{"arxiv_id":"2404.16821","doi":"10.48550/arxiv.2404.16821","metadata_source":"pith","pith_arxiv_id":"2404.16821","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","venue":"cs.CV","work_id":"3714835e-c5a6-4d7e-950c-be44670ed9e6","year":2024},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2404.16821","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:218e1ecc91b6e0314e9acf912c60cece4a849a2c9efb19ef508bcb6aa4b22f28","observation_id":"15550b04-61ab-458a-b25b-6ecde2b8a29b","resolution":{"observed_at":"2026-05-20T06:20:36.332513Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning , year=","venue":null,"work_id":"30499253-d99d-49fa-af31-c3755a37dbe3","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:2aa0208846e3c29c58c81096ca98f744dd11e7fdef27d3ba812b6be48c671e73","observation_id":"2c4051a9-1a87-41fd-91a3-323ab187e8f8","resolution":{"observed_at":"2026-05-20T06:20:36.844003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"c03f2a1e-da4a-40a0-b026-aaae4b8791cd","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:ccea496b904d76f5a226de987f9b85ee436f5c8d00f046c48277c1269b3f9362","observation_id":"618385b2-e032-4a55-8429-5103f9c4efb3","resolution":{"observed_at":"2026-05-20T06:20:36.845997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"30943486-0dbc-4e9c-b61e-8658b10f8108","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:a26215e30a94ad7d69339fb0912313f5fbc85f451248d716ba7aa6109db7dd34","observation_id":"3343d62f-8f14-4329-9194-3f02a33b645f","resolution":{"observed_at":"2026-05-20T06:20:36.847987Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14100","last_updated":"2022-12-15T19:21:35Z","snapshot_observed_at":"2026-08-04T09:31:34.784365Z","submitted_at":"2022-05-27T17:03:38Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","version":5},"cited_work":{"arxiv_id":"2205.14100","doi":"10.48550/arxiv.2205.14100","metadata_source":"pith","pith_arxiv_id":"2205.14100","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","venue":"cs.CV","work_id":"45b0563f-7563-4f00-8da5-8be70ff803fb","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2205.14100","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:f1ba8f53237926c41ecba23498ffa34fe6dc37ba6b46da05eadee2b86bbb59ab","observation_id":"8280fb1e-c6eb-47c7-8150-d422a1d9cef2","resolution":{"observed_at":"2026-05-20T06:20:36.491574Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-25T07:53:51.898843+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T07:53:51.898843+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.06794","last_updated":"2023-06-05T17:55:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-14T17:24:07Z","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","version":4},"cited_work":{"arxiv_id":"2209.06794","doi":null,"metadata_source":"pith","pith_arxiv_id":"2209.06794","snapshot_observed_at":"2026-07-04T19:30:07.491958Z","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","venue":"cs.CV","work_id":"29921cff-29c1-4aad-a27d-adac346027ec","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2209.06794","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:673c23aa662d71c4074be3e87e692b0bf65dd926639a03e13e9c250a57374125","observation_id":"e95bc2cb-2ed3-4f09-9ea4-a17ae1c7323f","resolution":{"observed_at":"2026-05-20T06:20:36.568176Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"578ba7fd-e41d-4150-a4f8-dd0ed78e5b2e","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:1d983df6f62e6dbada304588c5825d8b49dedc66c7716e62626d808d0b1796d5","observation_id":"8a99aa96-9325-405c-9b22-6efa8d54da14","resolution":{"observed_at":"2026-05-20T06:20:36.849804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"8366ee7b-a3aa-4e6f-ae38-6775d93d7451","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:f3114ed15a53ac17b0274a47dd273428513e9ed5e17178e70338cd9623932108","observation_id":"6b8f8874-c2ce-4656-8578-bcd40876a4e9","resolution":{"observed_at":"2026-05-20T06:20:36.851737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T04:40:41.097908Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"a675d477-b40a-4944-b8a6-2733c94587e8","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d42e870dcbf7001a203edc6ccd03b28bcaaec80632185907459035ae4b1fadff","observation_id":"b68542f5-6f08-47b1-ae87-63610d78ca0a","resolution":{"observed_at":"2026-05-20T06:20:36.854197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"7c9be260-b396-4913-8495-7b9264c28837","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d2533f8ea35da20d96129a60cb731e7cbeea247df78526f897a8e83573ee7f37","observation_id":"d3c1fb7a-e4b3-4b61-a5a9-e3ea664d1690","resolution":{"observed_at":"2026-05-20T06:20:36.856187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"7f6259a7-0962-4778-8ae1-4bd05ad78df5","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:2882f8d9be19459fff3d3394ef6d58b868a34a06aa46b29eea1c84d07a74116a","observation_id":"60c055f4-18a6-4f37-82f7-f6087c6bb65d","resolution":{"observed_at":"2026-05-20T06:20:36.865122Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"7abd768a-4878-4113-9089-248f4ab4d244","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:7914a128cf55a5735a87f067bf616d2055fde6cb17ce87b760a5337efffb28fa","observation_id":"173a94b3-db98-48ef-93d6-96fc96bc7b4d","resolution":{"observed_at":"2026-05-20T06:20:36.867260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.12561","last_updated":"2023-06-06T00:28:34Z","snapshot_observed_at":"2026-08-05T19:43:10.180125Z","submitted_at":"2022-11-22T20:26:44Z","title":"Retrieval-Augmented Multimodal Language Modeling","version":2},"cited_work":{"arxiv_id":"2211.12561","doi":"10.48550/arxiv.2211.12561","metadata_source":"arxiv_reference","pith_arxiv_id":"2211.12561","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Retrieval-augmented multi- modal language modeling","venue":"arXiv (Cornell University)","work_id":"1ff8e5a2-6229-4da5-9db4-e2ea51fae1a1","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2211.12561","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:833e5d17816e07fa87b6545470ef77a36d932e18058a5b30fa2eac49c2fa512c","observation_id":"a442456b-ee74-40b2-b22b-aa57346c2b3b","resolution":{"observed_at":"2026-05-20T06:20:36.381184Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision , pages=","venue":null,"work_id":"27380eed-0540-487a-8152-6b94652627da","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:5df088a13e589f8df2ec76a6ef1d052c4b45482a626164f2574a4e1bc8aeef74","observation_id":"f3bf0015-a5bd-4cc5-b815-455984531663","resolution":{"observed_at":"2026-05-20T06:20:36.869372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NeurIPS , year =","venue":null,"work_id":"dfd15358-e485-4037-8db2-e71d4b4caa55","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:85c5c7fd5fc1466be9fdfa5fb6a1b2c53971838647155156cfe722efdfeeb1d7","observation_id":"fa2f54f5-b5d8-4d21-a342-ac3040be2aac","resolution":{"observed_at":"2026-05-20T06:20:36.871669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"b0533152-89cd-4a3b-bb3a-41317449fe0b","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:4397d8c5226654b1def6cffecf1ec164c180da9a0ab22c783e65ccf14263e481","observation_id":"c3954fdc-5170-456a-9141-5e821f801377","resolution":{"observed_at":"2026-05-20T06:20:36.873716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies , pages=","venue":null,"work_id":"3beda477-9fac-404a-a69e-1f0b162d59d8","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:bf13c95af97fdea4f721eeaf350ec1a4bbf9abf73faece3a1b5bdfff2a3192fa","observation_id":"61211925-b1ef-4910-bb97-21e40b3b16c2","resolution":{"observed_at":"2026-05-20T06:20:36.875856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"78a170de-7dc9-4164-b803-d1d95d2158cd","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:81207608c3ae9296d555374e5a6e88fda68f92b650366f8cfd3051a9c1bc5c7c","observation_id":"c9587af5-780b-40ea-82de-3e86e6d28b66","resolution":{"observed_at":"2026-05-20T06:20:36.877751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision , pages=","venue":null,"work_id":"fb6b0c2b-916a-4898-9f22-9f7c1a0afcdc","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:67cae0f415bd27aed5c6132e8fd83247e49f68356a57e6ca2d7df6366b016074","observation_id":"3972d5a9-9d91-4a5c-b1ef-47d199c8f63e","resolution":{"observed_at":"2026-05-20T06:20:36.879732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"9da7d7eb-de85-4ffc-857b-9faf084e8836","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:3e14e2dc1f9bc94ad677ea5e97147d1dae3e655cfca83fd6b0c46a648489652c","observation_id":"2f4d5ca9-07c4-4ec6-9ca1-ad65d68f4174","resolution":{"observed_at":"2026-05-20T06:20:36.882031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"2b2b2874-99ee-47f2-a546-118e0c16659c","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:44bf52f867ef68382f405a916b7194f99d3dd9dd33968f6e8242c8ebd575cc41","observation_id":"459734e9-3e66-47a7-acf6-9c60bbe997a1","resolution":{"observed_at":"2026-05-20T06:20:36.883935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"fe89c5ec-0e57-4084-809c-9919e3501f6d","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:c8baac2f6c8c865a45e8c802edcfd5c324c6cb61bf1b2af21533c7023542ab34","observation_id":"62173ff7-ba5e-47a1-b44b-738f8e06a365","resolution":{"observed_at":"2026-05-20T06:20:36.886284Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T21:31:30.016849Z","title":"European conference on computer vision , pages=","venue":null,"work_id":"de7ad339-0de7-4f1e-8608-9c70e23c21c3","year":2020},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:c6e6742d7822de1ead5499a35e4ce2e4d094fe3a035f8fd5a531408e7139e5e9","observation_id":"1fdd638d-6af2-4dca-a9b2-b360ba23d48c","resolution":{"observed_at":"2026-05-20T06:20:36.888866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":"2002.05202","doi":"10.48550/arxiv.2002.05202","metadata_source":"pith","pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLU Variants Improve Transformer","venue":"cs.LG","work_id":"17d0763c-1016-41ab-a478-478e890765eb","year":2020},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d53bd2671434533da2ba7697fd4b0a63c201f5f6abebde67afe42a70eba873f7","observation_id":"5e39adac-8c17-4a81-a131-ee9321220ea2","resolution":{"observed_at":"2026-05-20T06:20:36.397564Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:50:07.002485+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:50:07.002485+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"89934e5e-5fa9-483a-9621-0ace5eec67ab","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:71113c967d589cfe7fb44bda452d9ed65d672bbd1d0f7021f567c16dcdd29f8f","observation_id":"47ae2a54-4c68-4fe3-b7b7-8ebf33bfda54","resolution":{"observed_at":"2026-05-20T06:20:36.893584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2023 , eprint=","venue":null,"work_id":"ab4d654f-8e98-4bce-a720-815dc1cb44c7","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:660d4adfb757ca5c801f670e0487d9f0b98a08a1a38fcf4062b690bf334f0ae5","observation_id":"3f08ae79-f840-43a5-9061-abb794005b2d","resolution":{"observed_at":"2026-05-20T06:23:06.064036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"d7db6114-a137-4c9d-a609-10816d037784","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:0b737c85573523359d9f9f6159d725c4dacfe799742b5ead9d41847fcee696c8","observation_id":"6368b160-0a54-4b3a-95d9-fa3da198c1b1","resolution":{"observed_at":"2026-05-20T06:23:06.046833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14565","last_updated":"2024-03-19T22:53:25Z","snapshot_observed_at":"2026-08-06T22:33:34.254048Z","submitted_at":"2023-06-26T10:26:33Z","title":"Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning","version":4},"cited_work":{"arxiv_id":"2306.14565","doi":"10.48550/arxiv.2306.14565","metadata_source":"pith","pith_arxiv_id":"2306.14565","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning","venue":"cs.CV","work_id":"ba8e8164-e47f-42d6-83ad-696cb57ee79a","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2306.14565","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:6f68d233bde25509ebbfed36c109d64ecc6cf347984d9e6fc21ae626526ca2fd","observation_id":"3ed11c71-f922-4ce9-839c-e6f1bc8540e0","resolution":{"observed_at":"2026-05-20T06:20:36.356515Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.04087","last_updated":"2023-12-28T16:01:36Z","snapshot_observed_at":"2026-08-06T02:13:50.627750Z","submitted_at":"2023-07-09T03:25:14Z","title":"SVIT: Scaling up Visual Instruction Tuning","version":3},"cited_work":{"arxiv_id":"2307.04087","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2307.04087","snapshot_observed_at":"2026-07-02T15:07:03.935337Z","title":"Svit: Scaling up visual instruction tuning","venue":null,"work_id":"cb1fc191-de60-4941-87ad-8829d88f83c2","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2307.04087","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:596b2588231bbdb3ef8e8a81b4c2969b31a4ae5bac699bc5e30fb301c7327f54","observation_id":"d1a283ce-0177-4dc8-a8da-bf383082fd9a","resolution":{"observed_at":"2026-05-20T06:20:36.368777Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T17:56:26.114328Z","title":"International conference on machine learning , pages=","venue":null,"work_id":"1031f0f0-cd67-4170-83ad-9a088b363c67","year":2021},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:f1d4b2e2413204707052107187c10e63d68936408963ba9c647db49d2551d6ba","observation_id":"e2317f9b-8599-4c53-8e34-11678bff397f","resolution":{"observed_at":"2026-05-20T06:23:06.034944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T16:46:21.684188Z","title":"Hashimoto , title =","venue":null,"work_id":"fadc9cdf-612b-4081-85df-c1904ccef4c8","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:89f311a2eac07471d14bfee841dd21b042bdda1df916ba1028a097d99539616b","observation_id":"87898014-2e2e-4545-b3f2-469bd9d4a018","resolution":{"observed_at":"2026-05-20T06:23:06.038535Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T04:46:46.761825Z","title":"2023 , publisher =","venue":null,"work_id":"8e6bfbff-0a0c-4ca7-81d7-23cd0ece3404","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d3f26487bb1fba9800b4cfeda9f8dc6ce92f2fca1bb937c8d498827c4ce6fdd3","observation_id":"ffb8abba-46ec-46d0-bee7-cee713c7fcb6","resolution":{"observed_at":"2026-05-20T06:23:06.033138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":"2306.13394","doi":"10.48550/arxiv.2306.13394","metadata_source":"pith","pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-07-10T13:27:05.561984Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","venue":"cs.CV","work_id":"806d2e73-71b3-4d56-87e0-39d571cc15d6","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:ec85d5f1adebef061a377800ebb26fc94488d8d68e859783f5841ad687eb0f4a","observation_id":"16dd2632-0ae1-4f92-983c-b1fa3a700b35","resolution":{"observed_at":"2026-05-20T06:20:36.446419Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16125","last_updated":"2023-08-02T08:02:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-30T04:25:16Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","version":2},"cited_work":{"arxiv_id":"2307.16125","doi":"10.48550/arxiv.2307.16125","metadata_source":"pith","pith_arxiv_id":"2307.16125","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","venue":"cs.CL","work_id":"23881ff0-b851-474c-8712-90744cc07a3a","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2307.16125","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:358903aa79d0471d669b62cb225e369179c0ddb66dea5ab984a165ce9688db6f","observation_id":"6276443a-ba48-4c65-a2aa-85382af71546","resolution":{"observed_at":"2026-05-20T06:20:36.488698Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Computer Vision--ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11--14, 2016, Proceedings, Part IV 14 , pages=","venue":null,"work_id":"252e1d5f-d98c-40b7-a6f7-0bbeb79686b9","year":2016},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d3bcc8e79b23cd380324390ef5d534d2437daa242475aceb072cc48d4b4143ec","observation_id":"5e0b7c72-b03d-47e5-913d-e571be8c4224","resolution":{"observed_at":"2026-05-20T06:23:06.067495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14181","last_updated":"2024-01-01T14:48:48Z","snapshot_observed_at":"2026-07-06T16:23:18.453624Z","submitted_at":"2023-09-25T14:43:43Z","title":"Q-Bench: A Benchmark for General-Purpose Foundation Models on Low-level Vision","version":3},"cited_work":{"arxiv_id":"2309.14181","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.14181","snapshot_observed_at":"2026-06-30T13:34:40.544825Z","title":"Q-bench: A benchmark for general-purpose foundation models on low-level vision","venue":null,"work_id":"5cc0bbca-7cce-4f8c-bd9d-7b44a68d9062","year":2024},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2309.14181","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:af2e2c360f9642be082f16cb61fe5c07f28275ddf15d98e29275aa629eca5445","observation_id":"ea24bf46-f79e-4581-8d74-e8261f784f22","resolution":{"observed_at":"2026-05-20T06:20:36.557416Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ArXiv , year=","venue":null,"work_id":"83ecbbd5-8184-45f7-b3e4-0490f2384cd9","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:dab47788a34200684b509538f91d41a2cff59fc0d7f5c09dfa7e7e5cceced952","observation_id":"2e455416-bbef-4168-ba29-0d1a98080757","resolution":{"observed_at":"2026-05-20T06:23:06.029652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":"2009.03300","doi":"10.48550/arxiv.2009.03300","metadata_source":"pith","pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Massive Multitask Language Understanding","venue":"cs.CY","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","year":2020},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:ca324a762af9eb9af5868c852e857060080cfca03ead7f64f6a3065e31fe8771","observation_id":"6fc3f590-22c1-4606-a51a-c12b083a6bca","resolution":{"observed_at":"2026-05-20T06:20:36.592244Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09261","last_updated":"2022-10-17T17:08:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-17T17:08:26Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","version":1},"cited_work":{"arxiv_id":"2210.09261","doi":"10.48550/arxiv.2210.09261","metadata_source":"pith","pith_arxiv_id":"2210.09261","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","venue":"cs.CL","work_id":"513eb205-04ca-4722-9a43-a74e8cbe7e85","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2210.09261","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:e92c3dfd1dd540d27af30a284ddf575627bc052a3b0a53a0a5e7cce491d5f1b5","observation_id":"60b36141-770e-44b0-8135-0e137ef45a1e","resolution":{"observed_at":"2026-05-20T06:20:36.319194Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06364","last_updated":"2023-09-18T14:23:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-13T09:39:30Z","title":"AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models","version":2},"cited_work":{"arxiv_id":"2304.06364","doi":"10.48550/arxiv.2304.06364","metadata_source":"pith","pith_arxiv_id":"2304.06364","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models","venue":"cs.CL","work_id":"d42c58d5-eeb0-462a-92ab-3081ee269e59","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2304.06364","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:325a03d945fe5909274b77751928945722b8a87ed1cbc23ba03526ce37662c2c","observation_id":"bc643947-5dc4-4ac4-b4fc-a6db1271f161","resolution":{"observed_at":"2026-05-20T06:20:36.324452Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":"1803.05457","doi":"10.1162/tacl_a_00448.https://aclanthology.org/2022.tacl-1.5","metadata_source":"pith","pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","venue":"cs.AI","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","year":2018},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:f6add47dc3b4ce8c34b6a4a973a83c77368015d056fe152c78a4b176040443d6","observation_id":"1a08c6b7-3c0b-491d-ac6c-dfa147dfaa7d","resolution":{"observed_at":"2026-05-20T06:20:36.328464Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the 25th ACM international conference on Multimedia , pages=","venue":null,"work_id":"47029572-d8cd-429c-b143-b241eed78783","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:1e3ed0d6e1aa7b667bf80778700f982b278a7a24c8e88da7d2eb289b3e53688e","observation_id":"0c51f8da-d84c-40c9-a7f9-ca643263d9f1","resolution":{"observed_at":"2026-05-20T06:23:06.028111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"ab677f8c-a774-4d0d-8ff0-9c1b7ee12039","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:73822eba196c40e7a70e3ca1dbc170e6410af2d1ebd0c2fec59bf832afab7ecb","observation_id":"a97c1128-9332-49b0-b54d-25c07fda91ec","resolution":{"observed_at":"2026-05-20T06:23:06.069417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2023 , eprint=","venue":null,"work_id":"f5c525cc-159b-4818-bf32-1ebe4e575d9c","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:98e231c0b77d9c759ddeb795e978df14be5d25b4fb014f852812dce32a8a2375","observation_id":"bd1fb5d2-f1ab-4256-aeb0-4db9e777d852","resolution":{"observed_at":"2026-05-20T06:23:06.018627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF international conference on computer vision , pages=","venue":null,"work_id":"c493e0ad-cbd3-41d1-9c0f-02295509cc74","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:0bb7804eec7ccad037b17c18d48dfa402003d4d6f772836c8dc15e9ac93a02cb","observation_id":"2364c3d2-e62b-4eb6-b278-d5b5eb496c17","resolution":{"observed_at":"2026-05-20T06:23:06.020696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The 2023 Conference on Empirical Methods in Natural Language Processing , year=","venue":null,"work_id":"bb968c60-b915-48c0-8dd4-11f509ae2d8d","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:a1b60342332632ca26ed29aa3ecaa11c8d5d330e4cf6fa2724988d2afdb0096a","observation_id":"aad7096f-a958-4ecd-aa90-85009d76a08f","resolution":{"observed_at":"2026-05-20T06:23:06.012599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T16:02:39.202588Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"dc9d036d-d357-4846-9592-a85d6480df86","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d9315048dd66b8913d09eea84d60e2dbec9b428a5ee7117dfb5f8d1a4dfc69ad","observation_id":"73483d60-7ded-49e4-9ff1-bc6852fde0c2","resolution":{"observed_at":"2026-05-20T06:23:06.016946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1da5317f-91a5-4219-af31-703a00ea947b","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:9d7f5c70e5654740e70e193ed153e55d628ff76290e214d3b5eb9d4328191b52","observation_id":"51412131-f3b4-4537-a344-7e3287006371","resolution":{"observed_at":"2026-05-20T06:23:06.010046Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":"b656a686-06e5-4028-b0f4-074a6fea3933","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:578f4bd7f0215a14e870cfa2e762b6d9afb78c0032950fd83066d8850db4612f","observation_id":"efe76cd9-42a9-44cf-87eb-0afb48a74dff","resolution":{"observed_at":"2026-05-20T06:23:06.042950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"ecaaddc6-dce3-4bf8-93f7-aa98be179716","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:9a4fddbf7e373cb68a5c0a3fd29cba5b7d205b9119a6e3ca9089d8e1b863e37f","observation_id":"111472cc-437a-42d0-8acf-13d335379d19","resolution":{"observed_at":"2026-05-20T06:23:06.026404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2022 , howpublished =","venue":null,"work_id":"62ba6819-e060-4f6d-adf1-ffb536f81eb9","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:6c9ad88c626d87cd6504c4e95ff25ca0a504c858f289d0cefaa6ac3ebe1e934e","observation_id":"dabce7bf-65da-4c23-8192-94447874a840","resolution":{"observed_at":"2026-05-20T06:23:06.024462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T18:01:21.619819Z","title":"Computer Vision--ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13 , pages=","venue":null,"work_id":"a6283ecf-c324-43f0-9469-453d23f90248","year":2014},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:d9f6f071923bb04b9feb85fa60f33253f95c814dee39a1faf0695f4e8c2c8eb7","observation_id":"b3db8ade-94a2-4f95-8e14-eba49653e079","resolution":{"observed_at":"2026-05-20T06:23:06.014593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Making the","venue":null,"work_id":"24b4130b-191b-404e-821a-a8d5cf32662d","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:0573bdadd221163cf4b7d787d41df805ec33146bc58e5a063eb619e3329b04d2","observation_id":"baf3f318-bcd0-4af5-8ffc-ac0bb0681d35","resolution":{"observed_at":"2026-05-20T06:23:06.078861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2019 international conference on document analysis and recognition (ICDAR) , pages=","venue":null,"work_id":"c53c35d3-1791-4921-8d13-d4f9569af1d0","year":2019},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:e75c2987dc3798f4ac02b6b031f4ea145b65efb8d713801b46eb727abb427bc1","observation_id":"e1b41dfd-6553-4f96-9e69-5dfc736d5adc","resolution":{"observed_at":"2026-05-20T06:23:06.075266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"96ee0538-13ad-455b-acc4-ae89bb0b0dcc","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:9973f072c6133c19955555f784b38158718e0a48869eb3c999a8211d25a26df5","observation_id":"fb3aafd5-6a05-491a-aeab-84cfdc61d901","resolution":{"observed_at":"2026-05-20T06:23:06.036857Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"df24e8e2-1014-464b-a122-cd1257d90670","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:dc69c050579189c78004866343e0d0aa12bee56d7ae29521be4678330242097d","observation_id":"f786beee-39e0-48f5-a98a-56abbe7797b1","resolution":{"observed_at":"2026-05-20T06:23:06.040906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part II 16 , pages=","venue":null,"work_id":"a2d0439d-3cb9-4a9a-8a58-1f06945cbbf0","year":2020},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:04526fce145c1921aa1722d9b1e0e8578dc34b0a73eb599e00d28d5e23006b01","observation_id":"7280a6c9-5301-46e3-8c2b-a664cfe3307a","resolution":{"observed_at":"2026-05-20T06:23:06.048869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/cvf conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"6634ad7c-7c68-474d-9183-593fd7912179","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:1509ff719deaaa627d056caa0c304236742849f7d7bd14be17eff01a62f70a9a","observation_id":"0e2108c9-3e57-4226-b915-b74f688589f2","resolution":{"observed_at":"2026-05-20T06:23:06.073532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"European Conference on Computer Vision , pages=","venue":null,"work_id":"ea9d8218-fb96-40a7-872a-80e77e9a61a3","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:ffd2635ee852765adc03a7511b37115def5187f58a2ad0f44b6a24c488c7bd3c","observation_id":"9fc45383-eca4-41c8-b4b4-555684d65f86","resolution":{"observed_at":"2026-05-20T06:23:06.050596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International journal of computer vision , volume=","venue":null,"work_id":"c1c20d84-c968-4aeb-9ddb-a30597e28f1c","year":2017},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:5d34d187f3491d65019f2ef88517374dc966c32d0f6241970746d24cc060dfa2","observation_id":"27a7f066-5dac-44d7-8fd9-15d65a60fc2e","resolution":{"observed_at":"2026-05-20T06:23:06.007756Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Computer Vision--ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part II 14 , pages=","venue":null,"work_id":"cf4a6220-136a-43fc-b3a4-c7224b63b2aa","year":2016},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:cd74c4a0c72fa97eb436d2f49dfd208ddfa7b9383e0622fed8e4493693490c59","observation_id":"9d0133a3-e6f3-47de-834e-3c41370938d0","resolution":{"observed_at":"2026-05-20T06:23:06.077201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.04790","last_updated":"2023-06-13T13:31:12Z","snapshot_observed_at":"2026-07-06T15:24:38.226044Z","submitted_at":"2023-05-08T15:45:42Z","title":"MultiModal-GPT: A Vision and Language Model for Dialogue with Humans","version":3},"cited_work":{"arxiv_id":"2305.04790","doi":"10.48550/arxiv.2305.04790","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.04790","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Multimodal-gpt: A vision and language model for dialogue with humans","venue":"arXiv (Cornell University)","work_id":"e5fb1f2e-4ed2-454f-87a3-9e9c40f8fa31","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2305.04790","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:551293e373c3b625e36094284efbb7b6dfc68e1f0b47020bfdc8441c20c2f98a","observation_id":"4f0c569c-2947-40d3-b5a5-39c76f571b76","resolution":{"observed_at":"2026-05-20T06:20:36.388847Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T23:31:35.516699Z","title":null,"venue":null,"work_id":"b2fa25bb-75f7-4d60-b9a4-4bead50bd86e","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:cc7c06a9074bc27af33a48116732c98025d9ed80d628fbc310adb2a6bb0fe6ab","observation_id":"79617ae0-2c61-4ab2-84d6-3190b00c92c6","resolution":{"observed_at":"2026-05-20T06:23:06.005708Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.02131","last_updated":"2023-03-14T23:51:53Z","snapshot_observed_at":"2026-08-06T04:05:35.778035Z","submitted_at":"2022-08-03T15:11:01Z","title":"Masked Vision and Language Modeling for Multi-modal Representation Learning","version":2},"cited_work":{"arxiv_id":"2208.02131","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2208.02131","snapshot_observed_at":"2026-07-03T14:28:31.488306Z","title":"arXiv preprint arXiv:2208.02131 , year=","venue":null,"work_id":"b4673e89-f7a1-4bbd-ae3c-b66558968305","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2208.02131","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:a2714248196c7da66a7800f0f4ff4efc13f5ca3cc40d1ec0943a2ea1b320c4cd","observation_id":"55cf0588-1a63-43f3-a548-c69e5be1de70","resolution":{"observed_at":"2026-05-20T06:20:36.402341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":"4b4a5b57-4fd9-4718-8d62-44dfd665fd39","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:be468253a1988056d6a9ff47a10361d3c216ff9d6903277db4eca4be4fcf6f48","observation_id":"75f46906-3c13-4aca-a331-f74e13f0615c","resolution":{"observed_at":"2026-05-20T06:23:06.044909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1607.06450","last_updated":"2016-07-21T19:57:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-07-21T19:57:52Z","title":"Layer Normalization","version":1},"cited_work":{"arxiv_id":"1607.06450","doi":"10.1007/978-3-319-32025-0","metadata_source":"pith","pith_arxiv_id":"1607.06450","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Layer Normalization","venue":"stat.ML","work_id":"20a2d720-0046-4c7c-bcd6-327ec8143f69","year":2016},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/1607.06450","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:27469423f27adf79dacf17a0648209f0103cfa46a1a3f03a7868d6fd30400c51","observation_id":"6ae7e215-c939-4ac4-9d4c-1e01b59e066c","resolution":{"observed_at":"2026-05-20T06:20:36.449690Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":"2106.09685","doi":"10.4088/pcc.v03n0609","metadata_source":"pith","pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","venue":"cs.CL","work_id":"0426219a-789e-4964-adc8-a04538510818","year":2021},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:bef35fad67fc29b180a407371c27292206ba47594a9a3708a3f2989b19d18552","observation_id":"3c2de6fa-cf34-4987-ac84-5884a3a36d71","resolution":{"observed_at":"2026-05-20T06:20:36.461623Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":"1909.08053","doi":"10.48550/arxiv.1909.08053","metadata_source":"pith","pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","venue":"cs.CL","work_id":"c888e6d1-0b1d-43d6-9ef5-f0912a0efa1b","year":2019},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:85c76a97ec3b3400191c47d7240ab431d8038961299a91caa522b2fb7ccbd220","observation_id":"755bcfab-8b96-441c-98ba-e66b641fc83a","resolution":{"observed_at":"2026-05-20T06:20:36.485479Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T15:33:53.606060Z","title":"International Conference on Machine Learning , pages=","venue":null,"work_id":"f0ffc8f2-a885-4af2-8808-c3aa655855af","year":2022},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:920ee6d5703741ac8c44a6ff9349bc30c27007bfde0c7728b0d2712e883947db","observation_id":"a15194ec-c396-4ffe-b937-900484ea5728","resolution":{"observed_at":"2026-05-20T06:23:06.022490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T15:33:53.754499Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision , pages=","venue":null,"work_id":"a226172b-3320-4256-a547-def05cb98ef4","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:5d6bedb54c276d7f121c2ba8209f60d7d13295f5dbe68fc0c86e8c49946700dd","observation_id":"a6ad327f-dae7-4ff0-bbb7-30cc3804efa1","resolution":{"observed_at":"2026-05-20T06:23:06.031290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14800","last_updated":"2024-01-23T04:01:43Z","snapshot_observed_at":"2026-07-06T15:32:18.583350Z","submitted_at":"2023-05-24T06:52:47Z","title":"Exploring Diverse In-Context Configurations for Image Captioning","version":6},"cited_work":{"arxiv_id":"2305.14800","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.14800","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2305.14800 , year=","venue":null,"work_id":"8034a3f2-0f93-495d-869e-7e834f3efddf","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2305.14800","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:b95c837b5a9022364c54795c6cf0c374db3cfe605d37703cd2cfdde6c6034579","observation_id":"a3829c38-823c-4691-beb7-0f3f0998ebda","resolution":{"observed_at":"2026-05-20T06:20:36.495065Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01571","last_updated":"2023-12-04T02:03:23Z","snapshot_observed_at":"2026-08-04T15:17:47.832154Z","submitted_at":"2023-12-04T02:03:23Z","title":"How to Configure Good In-Context Sequence for Visual Question Answering","version":1},"cited_work":{"arxiv_id":"2312.01571","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.01571","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2312.01571 , year=","venue":null,"work_id":"f5d533ff-4f4b-4eb0-8a52-fc3d87a74e28","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2312.01571","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:76ea5a8420a8dcb858dbcbc6ef87321efeda63fa5f766e44c490e2ea97c4a32b","observation_id":"e66d72d4-c06f-42cb-83c9-5fdd9237b91b","resolution":{"observed_at":"2026-05-20T06:20:36.501279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10104","last_updated":"2024-10-31T03:02:43Z","snapshot_observed_at":"2026-07-06T17:03:06.967126Z","submitted_at":"2023-12-15T03:11:03Z","title":"Lever LM: Configuring In-Context Sequence to Lever Large Vision Language Models","version":4},"cited_work":{"arxiv_id":"2312.10104","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10104","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2312.10104 , year=","venue":null,"work_id":"597929fe-01ab-4da7-ba6f-46e1b0d990b7","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2312.10104","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:1d3af392e786ea2767d481337fd0c6b91d165198acac8405dd1886de43354c02","observation_id":"e1525ec3-824b-4709-8235-9f7ad85d1a7d","resolution":{"observed_at":"2026-05-20T06:20:36.515715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00351","last_updated":"2023-12-06T04:19:04Z","snapshot_observed_at":"2026-08-03T01:30:58.630796Z","submitted_at":"2023-12-01T04:57:20Z","title":"Manipulating the Label Space for In-Context Classification","version":2},"cited_work":{"arxiv_id":"2312.00351","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.00351","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2312.00351 , year=","venue":null,"work_id":"8b8c51ef-8f60-4a8c-a984-2a059dbbbd4f","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2312.00351","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:ee99b2e1d4515a684513a6c78bc94baca9125221197baf9972a3dc0968497635","observation_id":"0c0549af-ce0f-4f67-9497-b5d0257fbf9f","resolution":{"observed_at":"2026-05-20T06:20:36.528552Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17421","last_updated":"2023-10-11T05:07:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-29T17:34:51Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","version":2},"cited_work":{"arxiv_id":"2309.17421","doi":"10.48550/arxiv.2309.17421","metadata_source":"pith","pith_arxiv_id":"2309.17421","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","venue":"cs.CV","work_id":"344e9dbe-1d9b-4992-a4f1-9bc649978f46","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2309.17421","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:5d243bbb9cfbddcade53e86d351cc3761d1569f6b19851a01303fd7e155336d8","observation_id":"a297d3f5-7a6d-45a9-9dc4-69131e4c092d","resolution":{"observed_at":"2026-05-20T06:20:36.535304Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:7beefbb58e558c167de656a033234beda287b8ece391205c3a220dae2d44db40","observation_id":"faa4d1e7-e021-4d15-9c8d-f08cfd1dae1b","resolution":{"observed_at":"2026-05-20T06:20:36.538521Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10529","last_updated":"2024-01-25T04:11:57Z","snapshot_observed_at":"2026-07-06T17:17:44.177514Z","submitted_at":"2024-01-19T07:10:13Z","title":"Mementos: A Comprehensive Benchmark for Multimodal Large Language Model Reasoning over Image Sequences","version":2},"cited_work":{"arxiv_id":"2401.10529","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.10529","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mementos: A comprehensive benchmark for multimodal large language model reasoning over image sequences","venue":null,"work_id":"d8d29a08-f52e-451a-97b0-33636a800bc9","year":2024},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2401.10529","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:1c5f90769b8af913c02cc9d1364be71ee40e5e25e17aea8f033d11cb30aea8f8","observation_id":"04cca4ad-d10f-4e47-966c-7749221d48db","resolution":{"observed_at":"2026-05-20T06:20:36.548843Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T23:15:44.778861Z","title":"2023 , url=","venue":null,"work_id":"20acb4a6-31ba-4fe0-a7ef-9a3b4d6b2b96","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:9479e3a1545e2f2326e3e70c13b95cb4b43a5dc8659800f8be0e093e6ead0681","observation_id":"ee5ae40d-22fc-4835-b1ae-e9ab01209402","resolution":{"observed_at":"2026-05-20T06:23:06.061708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T20:52:55.287956Z","title":"LLaVA-NeXT: Improved reasoning, OCR, and world knowledge , url=","venue":null,"work_id":"e71d9099-0256-41d4-aa32-5c68d9b02eb8","year":null},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:4719de0f0ebec7a285b0131df0425788cab59e10c31c1d2c0d4f8dfd8655851c","observation_id":"b524ec34-9c4f-4248-bbcb-32bc0f777fee","resolution":{"observed_at":"2026-05-20T06:23:06.071464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":21,"parse_uncertain":0,"unresolved":3,"verified_exact":6,"verified_fuzzy":70},"total_outbound_references":238},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 238 outbound references and 56 inbound Pith citation observations for arXiv:2408.04840."}