{"as_of":"2026-08-04T22:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b9fb3e403f62ad27627a6066cd321c69e3fc6174204aacee1721f1bdb4dc41f2","coverage":[{"denominator":20,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-24T09:02:31.211260Z","state":"measured"},{"denominator":120,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":120,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":104,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T11:24:22.627249Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2303.16199","last_updated":"2024-09-18T23:54:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-28T17:59:12Z","title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","version":3},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-14T23:07:42.245641Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2303.16199"},"observation_digest":"sha256:dfafddcc9c48975129cfffa8a69cd005ca0e56c14092a6f7fc93279ed86b5e6c","observation_id":"76944607-a24b-4725-a4f2-3624f6bcd79a","resolution":{"observed_at":"2026-05-14T23:07:42.918442Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2305.10355","last_updated":"2023-10-26T02:52:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T16:34:01Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-11T13:44:09.626361Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2305.10355"},"observation_digest":"sha256:c7f8b02df49eedb6e75f5524a864941e2dea24beab2c58c81cbac7eba12fd25e","observation_id":"c3394408-acdc-493f-a1e5-9115a84c87b1","resolution":{"observed_at":"2026-05-11T13:44:09.893263Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:b5b455dc820f36051873b93b7c748700cb26da36c75b3ab0d4a5701e403bbfaf","observation_id":"b1aa540c-3827-4de9-bdb3-5caf9b589431","resolution":{"observed_at":"2026-05-16T02:56:42.609524Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2306.14565","last_updated":"2024-03-19T22:53:25Z","snapshot_observed_at":"2026-07-06T15:46:40.281717Z","submitted_at":"2023-06-26T10:26:33Z","title":"Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T17:34:56.836034Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2306.14565"},"observation_digest":"sha256:5e3fe62b849820313371e89ae09fa22583e69d7b26297b63cbc3be8db9257b0b","observation_id":"f8962a29-202a-4945-9ccc-b4868541b0ca","resolution":{"observed_at":"2026-05-14T17:34:56.988590Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2307.06281","last_updated":"2024-08-20T03:56:03Z","snapshot_observed_at":"2026-07-06T15:53:19.485466Z","submitted_at":"2023-07-12T16:23:09Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-12T17:20:53.687692Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2307.06281"},"observation_digest":"sha256:e717cf94595fa3eaad133daee548e302591cd9d9da3ed071d92658424cd560f8","observation_id":"153230d6-ba90-4b15-9dc5-2fa0aa48f79c","resolution":{"observed_at":"2026-05-12T17:20:53.880932Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2307.06435","last_updated":"2024-10-17T01:10:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-12T20:01:52Z","title":"A Comprehensive Overview of Large Language Models","version":10},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-19T20:28:38.900026Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2307.06435"},"observation_digest":"sha256:6ec60532f5fe21e29e2aac41e7d70c550f4c43a56b1b2068c89ac1003b19c415","observation_id":"6e22e92b-a58a-4578-a0df-435bbc5a0475","resolution":{"observed_at":"2026-05-19T20:28:39.264298Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2307.16125","last_updated":"2023-08-02T08:02:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-30T04:25:16Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T16:59:50.495335Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2307.16125"},"observation_digest":"sha256:c6309c33e86d1b9ec2a5b5729f9c05b186ddd6bb495b33e0b28ac9bddea0ad23","observation_id":"78427e1f-bc1a-40d0-8b56-b633d9b8f0dd","resolution":{"observed_at":"2026-05-12T16:59:50.618607Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2308.01390","last_updated":"2023-08-07T17:53:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-02T19:10:23Z","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-14T01:52:01.163900Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2308.01390"},"observation_digest":"sha256:d2d2602db9ddac62f2e19d8b19a9edafbb8d0992101a4a5310e158befed73405","observation_id":"5c0198fe-1053-450a-b3e8-9b545abb631d","resolution":{"observed_at":"2026-05-14T01:52:01.336399Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2309.01219","last_updated":"2025-09-14T09:34:46Z","snapshot_observed_at":"2026-07-06T16:13:46.112815Z","submitted_at":"2023-09-03T16:56:48Z","title":"Siren's Song in the AI Ocean: A Survey on Hallucination in Large Language Models","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T14:21:16.453610Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2309.01219"},"observation_digest":"sha256:5968bb7a1e17e9c922f40824696b33f2ea4fd39af19967d6bf4a65c9ddfb4376","observation_id":"45bac780-abc3-4ac0-b7a8-807648c83609","resolution":{"observed_at":"2026-05-12T14:21:16.530157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2309.14525","last_updated":"2023-09-25T20:59:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-25T20:59:33Z","title":"Aligning Large Multimodal Models with Factually Augmented RLHF","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-15T17:58:17.699042Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2309.14525"},"observation_digest":"sha256:90a9200451b3de7519b20b0e797f4728d6b4741ec153784a598f2f323f277f09","observation_id":"ff276869-2731-4eec-967e-10de06a5b520","resolution":{"observed_at":"2026-05-15T17:58:17.771669Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2309.17421","last_updated":"2023-10-11T05:07:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-29T17:34:51Z","title":"The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision)","version":2},"reference_index":146,"source":"pdf_text","source_observed_at":"2026-05-15T23:26:06.183574Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2309.17421"},"observation_digest":"sha256:351f9fbf97d4ba94a646ced302da502fe5077647a6bf279a1ae014ac354686fa","observation_id":"4a56a9e4-1bf6-4ad4-999a-014ee602264e","resolution":{"observed_at":"2026-05-15T23:26:06.485947Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2310.00754","last_updated":"2024-03-16T19:28:08Z","snapshot_observed_at":"2026-08-02T06:11:56.496482Z","submitted_at":"2023-10-01T18:10:53Z","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T22:46:52.791128Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2310.00754"},"observation_digest":"sha256:979e08942d9b63d5b49eafbd1707e71a7ebef14d3bafcec6eb9c7bc35c2cfdaf","observation_id":"ce33ff3c-5afd-4b5f-a530-f9ae1e4c4163","resolution":{"observed_at":"2026-05-17T22:46:52.820083Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2310.01852","last_updated":"2024-01-22T03:11:15Z","snapshot_observed_at":"2026-08-02T22:12:20.187464Z","submitted_at":"2023-10-03T07:33:27Z","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","version":7},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-17T03:27:58.952076Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2310.01852"},"observation_digest":"sha256:6630e21db839f135e30090c2cadacccf8587545e52f01f986bc6c838e73b45d0","observation_id":"dbef0d99-48d9-45d8-8792-bd82c02c9789","resolution":{"observed_at":"2026-05-17T03:27:59.111880Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-12T19:11:33.783746Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2310.03744"},"observation_digest":"sha256:c172f668c3f21aa0f50be1410de19b82ba4b1b42b529fd04222d1db65ea4fc5e","observation_id":"cd581fc1-a0c9-4997-9dd8-5fb8592363cf","resolution":{"observed_at":"2026-05-12T19:11:33.968118Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2310.09478","last_updated":"2023-11-07T18:25:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-14T03:22:07Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-16T07:13:08.867745Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2310.09478"},"observation_digest":"sha256:77ac3f0c0a69a03fa38d662ccb7b171a55a498ade20530eef3546f6836920d84","observation_id":"870f95de-9c13-47aa-ae07-365bd2148e3c","resolution":{"observed_at":"2026-05-16T07:13:09.002100Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2310.14566","last_updated":"2024-03-25T06:05:24Z","snapshot_observed_at":"2026-08-02T21:20:28.501823Z","submitted_at":"2023-10-23T04:49:09Z","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","version":5},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-17T01:22:04.035994Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2310.14566"},"observation_digest":"sha256:9c9476383725ab171e26e1238ca06ef05c652633a48844bddd75e97c69190c2f","observation_id":"4c4fd9db-3542-404d-b6c1-39921caeb41f","resolution":{"observed_at":"2026-05-17T01:22:04.174548Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.04257","last_updated":"2023-11-09T01:56:51Z","snapshot_observed_at":"2026-07-06T16:44:23.357719Z","submitted_at":"2023-11-07T14:21:29Z","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-18T03:18:51.582340Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.04257"},"observation_digest":"sha256:c6f12608ec1d917875821fd8a86be61487725b2991639a7e67cc725f5f8630f5","observation_id":"15a88951-f765-42ba-bef5-0210290d1b26","resolution":{"observed_at":"2026-05-18T03:18:51.778085Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.07575","last_updated":"2023-11-13T18:59:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-13T18:59:47Z","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-17T03:03:26.723464Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.07575"},"observation_digest":"sha256:aa6f5f2d320ddac36503c4e2b0bc14f9dffc67f72e013770dedd426d47e3cb60","observation_id":"88b2bf15-26bd-46fa-9b0f-4474af0248a8","resolution":{"observed_at":"2026-05-17T03:03:26.796658Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-14T18:08:01.166072Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.10122"},"observation_digest":"sha256:d645cbd9083252786412f9b6ec2b2defc0d8e6547e5ed2690c222a4a2e403818","observation_id":"4358022d-95c1-45e0-ab53-c662ce04dc35","resolution":{"observed_at":"2026-05-14T18:08:01.386169Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-04T08:17:54.774738Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-13T17:08:12.727773Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.12793"},"observation_digest":"sha256:54950af8eae79174948956aa92ae57da3d24e5d5d093ddf99c6d2c5cb365d769","observation_id":"f8fd3119-352a-4101-9d27-eaa6f1784838","resolution":{"observed_at":"2026-05-13T17:08:12.985200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.12871","last_updated":"2024-05-09T17:35:44Z","snapshot_observed_at":"2026-08-02T11:57:50.333488Z","submitted_at":"2023-11-18T01:21:38Z","title":"An Embodied Generalist Agent in 3D World","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T14:22:18.606817Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.12871"},"observation_digest":"sha256:68e3a7f9c49a8f0de6f1265018b6ef0ae5a9c980f3072bd4b14ce8edac6f48b1","observation_id":"1827c624-c1f7-4e8f-a474-d7470f1bfd18","resolution":{"observed_at":"2026-05-17T14:22:18.658537Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.16502","last_updated":"2024-06-13T15:02:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-27T17:33:21Z","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","version":4},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-15T05:37:41.401736Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.16502"},"observation_digest":"sha256:af608351e871af386a097d83976452ee2353b117fe9761d4fe2e7677f531ab7c","observation_id":"6370e879-f16f-40d3-9cd9-0a591343968a","resolution":{"observed_at":"2026-05-15T05:37:41.704687Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2311.17005","last_updated":"2024-05-23T14:49:29Z","snapshot_observed_at":"2026-07-06T16:53:55.269172Z","submitted_at":"2023-11-28T17:59:04Z","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","version":4},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-17T20:22:34.954228Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2311.17005"},"observation_digest":"sha256:ed1b2ca4a8cf4e7799263d4be4036e08e5450d56fcd65acdfd26a33bcf395faa","observation_id":"54661be6-1b54-4002-b669-da2f5e91851d","resolution":{"observed_at":"2026-05-17T20:22:35.123593Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2312.16886","last_updated":"2023-12-30T04:59:21Z","snapshot_observed_at":"2026-07-29T22:12:01.108911Z","submitted_at":"2023-12-28T08:21:24Z","title":"MobileVLM : A Fast, Strong and Open Vision Language Assistant for Mobile Devices","version":2},"reference_index":127,"source":"pdf_text","source_observed_at":"2026-05-16T16:35:37.937462Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2312.16886"},"observation_digest":"sha256:73c4963042263e7be64adf84d0d8729122d21dace18401ff7f5504e41c8ca287","observation_id":"612cebfb-9871-416e-b028-61034679cc2d","resolution":{"observed_at":"2026-05-16T16:35:38.171663Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2401.10935","last_updated":"2024-02-23T04:36:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-17T08:10:35Z","title":"SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents","version":2},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-17T10:09:46.447508Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2401.10935"},"observation_digest":"sha256:5d7ff2af0c0bf6bd9eeb0153bf54e5a305ce7299956497e6dc4a3f4622cde6d2","observation_id":"25c59aa5-2992-4d29-85d6-5b5b6a0230c6","resolution":{"observed_at":"2026-05-17T10:09:46.632276Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T02:33:30.143907Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2401.15947"},"observation_digest":"sha256:b8522a6bae3565aa76c684faa6db9beaec49e2cecc0cd0f415067c4b8166739b","observation_id":"3c38bf03-f69f-42d8-87d8-fd248438338d","resolution":{"observed_at":"2026-05-16T02:33:30.430063Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2401.16158","last_updated":"2024-04-18T06:53:38Z","snapshot_observed_at":"2026-07-06T17:21:51.054477Z","submitted_at":"2024-01-29T13:46:37Z","title":"Mobile-Agent: Autonomous Multi-Modal Mobile Device Agent with Visual Perception","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T00:19:27.965902Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2401.16158"},"observation_digest":"sha256:cf94dcb069eea6f39b0c81e10a26dfe4e8e92442c67510d6a1336ee6605d4199","observation_id":"29f21e54-a2f2-4286-9570-f57881e28d40","resolution":{"observed_at":"2026-05-17T00:19:28.009730Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2402.03766","last_updated":"2024-02-06T07:16:36Z","snapshot_observed_at":"2026-07-06T17:26:02.434883Z","submitted_at":"2024-02-06T07:16:36Z","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-18T15:27:51.839171Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2402.03766"},"observation_digest":"sha256:13b4df9b65d6821959ee9db1f655c074b1c5298c77bd66156e7fb351324279ed","observation_id":"e779d337-483a-43dd-9583-b1117bc6f29d","resolution":{"observed_at":"2026-05-18T15:27:52.125475Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2402.11411","last_updated":"2024-02-18T00:56:16Z","snapshot_observed_at":"2026-07-06T17:31:41.687799Z","submitted_at":"2024-02-18T00:56:16Z","title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","version":1},"reference_index":179,"source":"arxiv_source","source_observed_at":"2026-05-17T10:58:53.215887Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2402.11411"},"observation_digest":"sha256:599a1f096ee51aca4efe65d3643a03e1fcad280fad5449890d9c16eaf858dbfd","observation_id":"f4ede3bb-da84-41ef-a5aa-b6de47c2bc2b","resolution":{"observed_at":"2026-05-17T10:58:53.480487Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-05-17T02:46:16.632743Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2403.00476"},"observation_digest":"sha256:d18f325f8d6786a450eec5f9528fc9e88fd74b375f4e603ed3e18f6c79a057d6","observation_id":"c66988cc-63eb-421a-aa47-81bf253b1b29","resolution":{"observed_at":"2026-05-17T02:46:16.862645Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"reference_index":124,"source":"pdf_text","source_observed_at":"2026-05-16T04:09:36.019146Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2403.09611"},"observation_digest":"sha256:849a7a8318c5d7153df798ae9beddb65e806a0272b31c0cd60865a489ddadb61","observation_id":"ffea0252-1c39-4f0f-961d-a817d10e3b08","resolution":{"observed_at":"2026-05-16T04:09:36.359145Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2403.10559","last_updated":"2026-04-21T03:03:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T06:51:26Z","title":"Generative Models and Connected and Automated Vehicles: A Survey in Exploring the Intersection of Transportation and AI","version":4},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-05-24T02:48:10.934475Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2403.10559"},"observation_digest":"sha256:07aae00e016533c1d94f0d2a3d234afce8883dcc52b4c24105141ed81b78a7bc","observation_id":"4acff808-ec0f-4638-b458-70db110dbda5","resolution":{"observed_at":"2026-05-24T02:48:47.383923Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2403.20330","last_updated":"2024-04-09T15:17:50Z","snapshot_observed_at":"2026-08-02T21:55:15.857183Z","submitted_at":"2024-03-29T17:59:34Z","title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-12T19:41:44.263663Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2403.20330"},"observation_digest":"sha256:0c7b54931d8627419016eb42c5d967a89ccbe6c4ff6b59179ad61e8c56d1e1d2","observation_id":"fe117760-de67-47c6-86df-6ba21ed3c357","resolution":{"observed_at":"2026-05-12T19:41:44.416241Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2404.18930","last_updated":"2025-04-01T18:36:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-29T17:59:41Z","title":"Hallucination of Multimodal Large Language Models: A Survey","version":2},"reference_index":188,"source":"pdf_text","source_observed_at":"2026-05-11T12:33:32.631346Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2404.18930"},"observation_digest":"sha256:8afb477a886e69473969934063c5a41751daaf9e871cf820bb8af71cc28385b7","observation_id":"0db715e4-c6ce-4d06-935a-ecb9cfbe959a","resolution":{"observed_at":"2026-05-11T12:33:33.243343Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2406.03520","last_updated":"2024-10-03T17:24:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-05T17:53:55Z","title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","version":2},"reference_index":114,"source":"pdf_text","source_observed_at":"2026-05-20T11:34:37.599691Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2406.03520"},"observation_digest":"sha256:8501ff81ec175a0d79671bf913a5f3ab45a71cf98b1bb6fc241618a45c407c5e","observation_id":"43f69a16-3446-4395-b26d-a6b62830dca0","resolution":{"observed_at":"2026-05-20T11:34:37.739196Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:127f92e04ac65c0d60ae39786dc5759e139a28c206f247ed52281639404980d2","observation_id":"2fca4937-5447-4c41-af00-8ddddb1a404f","resolution":{"observed_at":"2026-05-14T19:55:26.482600Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-11T02:44:53.284345Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2406.07476"},"observation_digest":"sha256:d27a78a4ff72b9616f0169521b61804ef47c9994cba1daf78df813e66aa393c1","observation_id":"d5c109a0-9114-462a-bf25-31dd5942c855","resolution":{"observed_at":"2026-05-11T02:44:53.704805Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2407.01284","last_updated":"2024-07-01T13:39:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-01T13:39:08Z","title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T21:55:40.808698Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2407.01284"},"observation_digest":"sha256:dbb64419ced49e93911af19f2be59021eadf95b888e508645224c9c60411459a","observation_id":"b7089374-dc6c-4598-9382-6d94c5bedff9","resolution":{"observed_at":"2026-05-16T21:55:41.004026Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":261,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:05852150f94243ee11ee03db7c2337e7ee86208bf2e1aa66af3390897e234c8d","observation_id":"14aa7ede-5221-4739-83d3-02b5cd04641a","resolution":{"observed_at":"2026-05-20T06:20:36.560997Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:cfb83926468c4a1b99d6c2883ad5e9fd263eba7e639feeda19b3555600353467","observation_id":"a9bdbae0-afc3-4e40-ad65-3f37fe8d9869","resolution":{"observed_at":"2026-05-16T07:59:32.734287Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2409.02813","last_updated":"2025-05-22T08:22:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-04T15:31:26Z","title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","version":3},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-14T00:51:48.163349Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2409.02813"},"observation_digest":"sha256:317c0d515034b1d203cafa0900302d6e945fcd60db79bfe8d9c942afdeb25b94","observation_id":"e64a57d2-caf1-41e1-b96a-46707c56b9e6","resolution":{"observed_at":"2026-05-14T00:51:48.441410Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2410.04417","last_updated":"2025-06-03T04:12:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-06T09:18:04Z","title":"SparseVLM: Visual Token Sparsification for Efficient Vision-Language Model Inference","version":4},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-15T14:58:32.303101Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2410.04417"},"observation_digest":"sha256:cc71d1c186372db3eae6c6066e884b4859ef18bff26b8e37393e0aafcb3d6b58","observation_id":"dca7e261-89ab-422a-802c-6b3872e4f7e1","resolution":{"observed_at":"2026-05-15T14:58:32.359216Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2410.14702","last_updated":"2026-05-10T19:30:43Z","snapshot_observed_at":"2026-07-06T19:36:10.759412Z","submitted_at":"2024-10-06T20:35:41Z","title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-23T20:03:38.336841Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2410.14702"},"observation_digest":"sha256:8ab58042db9acd637571a68248bd87390595830e8c8a74a4fd0809ca1003d373","observation_id":"3bb13275-a751-4d4e-a841-41ade18e559f","resolution":{"observed_at":"2026-05-23T20:05:47.864373Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T13:53:33.585035Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2410.17434"},"observation_digest":"sha256:c4be5137b9309d5ad41e8d5bceb9b1c2ed57a44cebea2de4a0fcb5329bda0df9","observation_id":"584107d8-04e9-44bc-8705-32a18bcb5e03","resolution":{"observed_at":"2026-05-16T13:53:33.714961Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2411.16771","last_updated":"2026-04-23T13:21:19Z","snapshot_observed_at":"2026-08-02T03:20:43.405143Z","submitted_at":"2024-11-25T06:17:23Z","title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:12.821916Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2411.16771"},"observation_digest":"sha256:0448992fbac41548aaeecdef010044b1705dccd7ab6b04ebe1ea56e967724fc9","observation_id":"59dd7a6c-167a-454e-84d6-48c4e8adf649","resolution":{"observed_at":"2026-05-23T16:58:12.147855Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2411.18111","last_updated":"2026-05-11T16:39:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-27T07:45:25Z","title":"When Large Vision-Language Models Meet Person Re-Identification","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-23T16:58:02.417053Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2411.18111"},"observation_digest":"sha256:8d54ab9f2f021dd41afd5764608530aa693e4ab04a2b4ddc44c6a6c3a66ca4e4","observation_id":"5932dfb7-3179-4cc2-a915-3303f017e8d2","resolution":{"observed_at":"2026-05-23T16:58:11.858648Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2412.00131","last_updated":"2024-11-28T14:07:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-28T14:07:45Z","title":"Open-Sora Plan: Open-Source Large Video Generation Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T08:38:27.946746Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2412.00131"},"observation_digest":"sha256:4b0b90e09a5d91ba141c34e83b4750171a1f51911663354148c0e7585a3113a4","observation_id":"fa1e2b3a-1aca-45ce-bfec-f549f0c2e0ff","resolution":{"observed_at":"2026-05-23T08:42:45.159395Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2412.17574","last_updated":"2026-04-13T15:05:36Z","snapshot_observed_at":"2026-07-06T20:12:09.120832Z","submitted_at":"2024-12-23T13:45:56Z","title":"HumanVBench: Probing Human-Centric Video Understanding in MLLMs with Automatically Synthesized Benchmarks","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-23T07:05:08.716223Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2412.17574"},"observation_digest":"sha256:a29db0d162ca3a2e7f4d5f2480210908099a767f69488ebcd67e120a955dd0c5","observation_id":"e3eb39e2-a421-42ba-801a-86a9868077ed","resolution":{"observed_at":"2026-05-23T07:05:29.213504Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:bbd0e2a97b535be30410637d9fe9790dc435e823f2820b62f17d7a1861c84089","observation_id":"0558c787-2bf6-4a06-9024-203621515970","resolution":{"observed_at":"2026-05-23T06:02:37.657152Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2503.07536","last_updated":"2025-03-11T03:32:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-10T17:04:14Z","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-16T15:15:46.255296Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2503.07536"},"observation_digest":"sha256:3e4b50abc894b992cda2471a2ed4df23d3ee4db0984ea1492d743b19f5966ff4","observation_id":"9e60d85c-99e4-4327-829d-5946a85e52b5","resolution":{"observed_at":"2026-05-16T15:15:46.338926Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2504.07148","last_updated":"2026-04-13T07:27:05Z","snapshot_observed_at":"2026-08-03T08:08:22.792118Z","submitted_at":"2025-04-09T03:12:44Z","title":"Q-Agent: Quality-Driven Chain-of-Thought Image Restoration Agent through Robust Multimodal Large Language Model","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-22T21:17:15.694349Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2504.07148"},"observation_digest":"sha256:991f7cb9f64ce751d02792b3a99587b794460c9b917d3ee54b8689ffe58888c9","observation_id":"8a2f37d5-f00c-4f4e-8bec-9a88201beeee","resolution":{"observed_at":"2026-05-22T21:22:09.793387Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2505.21472","last_updated":"2025-08-11T19:24:13Z","snapshot_observed_at":"2026-07-06T21:31:35.572708Z","submitted_at":"2025-05-27T17:45:21Z","title":"Mitigating Hallucination in Large Vision-Language Models via Adaptive Attention Calibration","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-19T12:48:44.324236Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2505.21472"},"observation_digest":"sha256:dba0774069b678d45344f34b407f2001b179693d65f95eba48cd8882e1689406","observation_id":"d9029727-9535-4217-990a-46e0d006d43e","resolution":{"observed_at":"2026-05-19T12:52:18.070105Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2506.09522","last_updated":"2026-05-13T05:18:27Z","snapshot_observed_at":"2026-07-06T21:40:17.050453Z","submitted_at":"2025-06-11T08:46:55Z","title":"Revisit What You See: Revealing Visual Semantics in Vision Tokens to Guide LVLM Decoding","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T09:42:14.171289Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2506.09522"},"observation_digest":"sha256:8c2be059fc7f3c68edf280e2a06f701a1bf06e7f4b31c50192c42587f2f2e162","observation_id":"ee130942-ea7a-4237-81b9-e3f9d0b59eb5","resolution":{"observed_at":"2026-05-19T09:43:02.172931Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2507.09861","last_updated":"2026-04-21T13:31:05Z","snapshot_observed_at":"2026-07-06T21:56:35.665677Z","submitted_at":"2025-07-14T02:10:31Z","title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-19T04:38:49.512293Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2507.09861"},"observation_digest":"sha256:783c451466c68f97b1c8e3e4210c661b3bdcbbc81e57897c661c0c6dbc4b7dcf","observation_id":"bc6634c1-5380-4f0c-bbe8-421a3de6dab4","resolution":{"observed_at":"2026-05-19T04:42:04.355425Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2509.15435","last_updated":"2026-05-19T14:30:01Z","snapshot_observed_at":"2026-07-06T22:30:17.163909Z","submitted_at":"2025-09-18T21:17:23Z","title":"ORCA: An Agentic Reasoning Framework for Hallucination and Adversarial Robustness in Vision-Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-18T15:23:13.318310Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2509.15435"},"observation_digest":"sha256:cdb4fcdbae17f231200118049e4ba401bb8e5ee4e578a975f1953b7b51373d3a","observation_id":"575aeedd-6a0d-48fb-a16f-3af8b3d3eacb","resolution":{"observed_at":"2026-05-18T15:26:33.887861Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2509.15435","last_updated":"2026-05-19T14:30:01Z","snapshot_observed_at":"2026-07-06T22:30:17.163909Z","submitted_at":"2025-09-18T21:17:23Z","title":"ORCA: An Agentic Reasoning Framework for Hallucination and Adversarial Robustness in Vision-Language Models","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T21:28:59.898029Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2509.15435"},"observation_digest":"sha256:a9341178906dedc320ff257adc848b08db86a9f0ed864803fd35fa9e86fe04f3","observation_id":"656eb6dd-2e8a-4b3e-a877-200dba689f01","resolution":{"observed_at":"2026-05-21T21:30:39.467615Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-04T11:24:22.627249Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.05544","last_updated":"2026-06-03T18:20:52Z","snapshot_observed_at":"2026-08-04T16:54:48.294180Z","submitted_at":"2025-10-07T03:07:47Z","title":"Activation-Informed Pareto-Guided Low-Rank Compression for Efficient LLM/VLM","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-04T11:24:22.627249Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2510.05544"},"observation_digest":"sha256:fd8cc66d74aff10c3830a52947eb9fb0980758a38fcd5ca03f6cd09874a4b52b","observation_id":"6f18ac0c-2eff-4346-8fd9-94fb847f3974","resolution":{"observed_at":"2026-08-04T11:24:22.627249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-04T10:18:44.550010Z","title":"Peter Young, Alice Lai, Micah Hodosh, and Julia Hockenmaier","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.10921","last_updated":"2026-05-25T02:35:42Z","snapshot_observed_at":"2026-08-04T10:18:40.408381Z","submitted_at":"2025-10-13T02:32:07Z","title":"FG-CLIP 2: A Bilingual Fine-grained Vision-Language Alignment Model","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T10:18:44.550010Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2510.10921"},"observation_digest":"sha256:536ed723eedaba770eec88c89f4a2cc02f6b24da98bd22d24d0ab813825b024a","observation_id":"0981b908-739f-4a9f-aa52-001c61bf3bca","resolution":{"observed_at":"2026-08-04T10:18:44.550010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-03T20:23:08.346069Z","title":"mplug-owl: Modularization empowers large language models with multimodality.arXiv preprint arXiv:2304.14178, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.20272","last_updated":"2026-07-03T06:27:17Z","snapshot_observed_at":"2026-08-03T20:22:57.328566Z","submitted_at":"2025-11-25T12:58:32Z","title":"VKnowU: Evaluating Visual Knowledge Understanding in Multimodal LLMs","version":2},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-03T20:23:08.346069Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2511.20272"},"observation_digest":"sha256:85c633d186fc32d497ae2c08355b96b2f862cb48aa49f3703d1b1eabcfb8255e","observation_id":"e59d0de6-02af-4d0f-98a8-6243b17f1a14","resolution":{"observed_at":"2026-08-03T20:23:08.346069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2512.03438","last_updated":"2026-07-30T22:27:17Z","snapshot_observed_at":"2026-08-04T21:10:06.214190Z","submitted_at":"2025-12-03T04:42:47Z","title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-17T03:09:31.161760Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2512.03438"},"observation_digest":"sha256:b27862a5c447dc9bef3f24421432367e661e1950772197ecd58ac2f7d01bbc8f","observation_id":"0cdc6667-9167-471e-ba50-2e699eb78d92","resolution":{"observed_at":"2026-05-17T03:11:29.873760Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-03T18:51:48.397328Z","title":"mplug-owl: Modularization empowers large language models with multimodality.arXiv preprint arXiv:2304.14178, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2512.03438","last_updated":"2026-07-30T22:27:17Z","snapshot_observed_at":"2026-08-04T21:10:06.214190Z","submitted_at":"2025-12-03T04:42:47Z","title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-03T18:51:48.397328Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2512.03438"},"observation_digest":"sha256:ab798e102d5c130c2984213cdf480f7a0fe5f7bebbfc2dab4571533e13e925b7","observation_id":"92116bc7-0121-4aed-a4df-7aac4060b207","resolution":{"observed_at":"2026-08-03T18:51:48.397328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-03T13:02:00.686182Z","title":"arXiv preprint arXiv:2304.14178 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.01095","last_updated":"2026-08-02T22:17:06Z","snapshot_observed_at":"2026-08-04T21:27:29.746970Z","submitted_at":"2026-01-03T07:12:55Z","title":"NarrativeTrack: Evaluating Entity-Centric Reasoning for Narrative Understanding","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-03T13:02:00.686182Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2601.01095"},"observation_digest":"sha256:e4463af70b4b23deae6a90d9b04b080105419073c0717cc814f2963c9c3b315e","observation_id":"8217daf5-0d1c-46b2-ac43-aa2968335a53","resolution":{"observed_at":"2026-08-03T13:02:00.686182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-04T06:36:33.949676Z","title":"arXiv preprint arXiv:2304.14178 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.01095","last_updated":"2026-08-02T22:17:06Z","snapshot_observed_at":"2026-08-04T21:27:29.746970Z","submitted_at":"2026-01-03T07:12:55Z","title":"NarrativeTrack: Evaluating Entity-Centric Reasoning for Narrative Understanding","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-04T06:36:33.949676Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2601.01095"},"observation_digest":"sha256:d5b361098f7e5076c3ce490dccf1694ae4bf6338bd3416c2d52ec582db94a917","observation_id":"60f37dde-6ba9-43ff-88da-749d9fe23481","resolution":{"observed_at":"2026-08-04T06:36:33.949676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-03T05:29:30.854898Z","title":"mplug-owl: Modularization empowers large language models with multimodality.arXiv preprint arXiv:2304.14178, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.02401","last_updated":"2026-07-07T06:51:43Z","snapshot_observed_at":"2026-08-03T05:29:25.861855Z","submitted_at":"2026-02-02T17:59:01Z","title":"Superman: Unifying Skeleton and Vision for Human Motion Perception and Generation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T05:29:30.854898Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2602.02401"},"observation_digest":"sha256:ad30aab26d393eee92edaa1811163261039f7266d3934cc64d53b346bbe9f295","observation_id":"6521beb8-f74b-476a-a313-c4b183aaa3e8","resolution":{"observed_at":"2026-08-03T05:29:30.854898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-15T13:43:40.241796Z","title":"mplug-owl: Modularization empowers large language models with multimodality.arXiv preprint arXiv:2304.14178, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.06577","last_updated":"2026-07-03T04:02:30Z","snapshot_observed_at":"2026-08-02T23:56:12.139811Z","submitted_at":"2026-03-06T18:59:57Z","title":"Omni-Diffusion: Unified Multimodal Understanding and Generation with Masked Discrete Diffusion","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-15T13:43:40.241796Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2603.06577"},"observation_digest":"sha256:ff151e9bf731789ff4090df2b7437ab4af262bd71d8a3fea9c86ddb8dd331109","observation_id":"67790b3b-d1f7-4a77-842c-a0b3b64531b6","resolution":{"observed_at":"2026-07-15T13:43:40.241796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-14T22:18:56.115559Z","title":"arXiv preprint arXiv:2304.14178 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.12478","last_updated":"2026-07-09T06:30:13Z","snapshot_observed_at":"2026-08-04T00:27:12.184147Z","submitted_at":"2026-03-12T21:54:50Z","title":"Less Data, Faster Convergence: Goal-Driven Data Optimization for Multimodal Instruction Tuning","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-07-14T22:18:56.115559Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2603.12478"},"observation_digest":"sha256:44af8e41c36d763dd53dff1a69099dcbc9b654a3e9258db51884997e42f0f725","observation_id":"6ab547cc-cda1-41e6-b117-eba8546c1d2b","resolution":{"observed_at":"2026-07-14T22:18:56.115559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2603.25120","last_updated":"2026-03-26T07:45:29Z","snapshot_observed_at":"2026-07-06T22:50:37.468743Z","submitted_at":"2026-03-26T07:45:29Z","title":"DFLOP: A Data-driven Framework for Multimodal LLM Training Pipeline Optimization","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-21T11:05:07.336827Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2603.25120"},"observation_digest":"sha256:fd074fc70f0ea63bd09673bfee8a0d2d10570efa32ef8ed13404d9adab3f770e","observation_id":"fa6d3f82-8639-42fa-8dd9-fcd97d668a2b","resolution":{"observed_at":"2026-05-21T11:10:02.352378Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:d137a0a102bcbed212743c09ed92fe7150f006ba15a4bb3bb967150f1c3563f7","observation_id":"4561c020-80f0-475e-b953-e97ae17bd251","resolution":{"observed_at":"2026-05-14T22:08:04.297157Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2603.27507","last_updated":"2026-04-26T20:14:44Z","snapshot_observed_at":"2026-07-06T22:50:55.897076Z","submitted_at":"2026-03-29T04:16:04Z","title":"Chat-Scene++: Exploiting Context-Rich Object Identification for 3D LLM","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-14T21:31:00.764078Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2603.27507"},"observation_digest":"sha256:56e734a94fdfc199625b99b7fc587f5200b09f6dab72ed2585d62b42ba2d72d9","observation_id":"603e4b0d-d928-43ff-96ac-d27b9c253815","resolution":{"observed_at":"2026-05-14T21:32:59.463336Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2604.03231","last_updated":"2026-04-03T17:59:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T17:59:51Z","title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-13T20:28:30.864143Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2604.03231"},"observation_digest":"sha256:53c2126b2d224dc97609f9ef9500c291d4eb95d19da7a2c970e966e7e9452e87","observation_id":"a23dda5a-a2fb-4b65-a15d-ee527e89e665","resolution":{"observed_at":"2026-05-13T20:33:17.188098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2604.11122","last_updated":"2026-04-13T07:36:02Z","snapshot_observed_at":"2026-07-06T22:59:36.571641Z","submitted_at":"2026-04-13T07:36:02Z","title":"Semantic-Geometric Dual Compression: Training-Free Visual Token Reduction for Ultra-High-Resolution Remote Sensing Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T15:29:49.681399Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2604.11122"},"observation_digest":"sha256:316599c1991e01711db5152cd128940e2ff7462fdf724dbf8ebaae85a3a31cf6","observation_id":"7a846a2b-d46c-4c2f-b105-de90b785b8a6","resolution":{"observed_at":"2026-05-11T10:26:01.395330Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2604.20696","last_updated":"2026-04-22T15:41:33Z","snapshot_observed_at":"2026-08-02T21:16:36.054502Z","submitted_at":"2026-04-22T15:41:33Z","title":"R-CoV: Region-Aware Chain-of-Verification for Alleviating Object Hallucinations in LVLMs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-10T01:28:14.039910Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2604.20696"},"observation_digest":"sha256:0be851661e2f9fd65e885a5454b40f90a1d9fcedc6ca6a6268e5100392c6d35f","observation_id":"d44ef7c9-740a-4b14-84b9-bbba089d64d9","resolution":{"observed_at":"2026-05-11T13:36:04.422982Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2604.20705","last_updated":"2026-04-22T15:46:42Z","snapshot_observed_at":"2026-07-06T23:07:25.185196Z","submitted_at":"2026-04-22T15:46:42Z","title":"SSL-R1: Self-Supervised Visual Reinforcement Post-Training for Multimodal Large Language Models","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-10T01:23:32.849326Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2604.20705"},"observation_digest":"sha256:278cce3f1374b1ec23942b73614511a8700c71a501cee888ddb6dbdb4d1edf4f","observation_id":"37c41524-cf77-4589-b7b7-e8f28b166e66","resolution":{"observed_at":"2026-05-11T13:36:08.832543Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2604.21343","last_updated":"2026-04-23T06:58:08Z","snapshot_observed_at":"2026-07-06T23:07:56.487654Z","submitted_at":"2026-04-23T06:58:08Z","title":"Latent Denoising Improves Visual Alignment in Large Multimodal Models","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-09T23:07:54.806529Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2604.21343"},"observation_digest":"sha256:29ab3447e1d3d1e6560844cdfee7fdb5072d2efd41b157777c2384fe7f478d9c","observation_id":"a0a9d8e0-8560-441b-9d16-5054bd9e2c72","resolution":{"observed_at":"2026-05-09T23:09:26.730864Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.05938","last_updated":"2026-08-03T04:01:00Z","snapshot_observed_at":"2026-08-04T21:28:04.244266Z","submitted_at":"2026-05-07T09:46:33Z","title":"ICU-Bench:Benchmarking Continual Unlearning in Multimodal Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T10:53:27.265815Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.05938"},"observation_digest":"sha256:f419e37a09190cb100067762582390f9c5aa9f774e0ee9ae420aeaf6ccc7f73d","observation_id":"ec175278-467b-4f7c-94d3-07433fceaa42","resolution":{"observed_at":"2026-05-11T19:51:11.023925Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-04T05:20:43.477913Z","title":"mplug-owl: Modularization empowers large language models with multimodality.arXiv preprint arXiv:2304.14178, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.05938","last_updated":"2026-08-03T04:01:00Z","snapshot_observed_at":"2026-08-04T21:28:04.244266Z","submitted_at":"2026-05-07T09:46:33Z","title":"ICU-Bench:Benchmarking Continual Unlearning in Multimodal Large Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T05:20:43.477913Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.05938"},"observation_digest":"sha256:8845accdfe44258929990d030aa401e435032520a117ed593317813c766427f9","observation_id":"72aaa357-9d65-4b0f-b511-d34c966df4e4","resolution":{"observed_at":"2026-08-04T05:20:43.477913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.06126","last_updated":"2026-05-07T12:31:22Z","snapshot_observed_at":"2026-08-03T10:59:40.481635Z","submitted_at":"2026-05-07T12:31:22Z","title":"AffectGPT-RL: Revealing Roles of Reinforcement Learning in Open-Vocabulary Emotion Recognition","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-08T07:25:42.870223Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.06126"},"observation_digest":"sha256:4613f900f92d58ea9929fa0b44bf6dfb21c3831b0e82fd0e403ea727e7116d3a","observation_id":"4a82033d-3cda-4f74-b2da-a22d1a1497ea","resolution":{"observed_at":"2026-05-11T21:01:11.833187Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.07477","last_updated":"2026-05-08T09:23:26Z","snapshot_observed_at":"2026-07-06T23:19:51.414344Z","submitted_at":"2026-05-08T09:23:26Z","title":"ReasonEdit: Towards Interpretable Image Editing Evaluation via Reinforcement Learning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-11T01:51:42.304051Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.07477"},"observation_digest":"sha256:9f589ac55aa3f975c70a68c197eb6eba894f5e8aba9fc2eaa8e7c95fb691672a","observation_id":"0ca23b14-1dcf-4089-ae60-7c869c778804","resolution":{"observed_at":"2026-05-11T04:15:59.717874Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.08985","last_updated":"2026-05-09T15:10:26Z","snapshot_observed_at":"2026-07-06T23:21:06.773761Z","submitted_at":"2026-05-09T15:10:26Z","title":"LLaVA-UHD v4: What Makes Efficient Visual Encoding in MLLMs?","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-12T02:06:45.858231Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.08985"},"observation_digest":"sha256:cc334cfbb1f0397b3b41404e69cd8a4b88a9633332fc77f96001759cc2bdfc80","observation_id":"a774b6b9-dc84-436d-85a5-29e93bdb4f59","resolution":{"observed_at":"2026-05-12T02:11:16.343060Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.11808","last_updated":"2026-05-12T09:03:19Z","snapshot_observed_at":"2026-07-06T23:23:34.924221Z","submitted_at":"2026-05-12T09:03:19Z","title":"Mitigating Action-Relation Hallucinations in LVLMs via Relation-aware Visual Enhancement","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-13T05:44:05.093926Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.11808"},"observation_digest":"sha256:bef92aff1cd103d3b3cdd7504300aefa3a1b06ec69cb14a237ca22308813169e","observation_id":"36c3ac5b-1ba3-4d49-a10e-b561e49192a7","resolution":{"observed_at":"2026-05-13T05:47:21.423502Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.16409","last_updated":"2026-05-13T14:16:39Z","snapshot_observed_at":"2026-07-06T23:27:34.514541Z","submitted_at":"2026-05-13T14:16:39Z","title":"Multilingual OCR-Aware Fine-Tuning and Prompt-Guided Chain-of-Thought Reasoning for Multimodal Large Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T21:21:11.388050Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.16409"},"observation_digest":"sha256:7974c3d4cdb987238aef088b5ff8ac13c51c8a0e5ff411070ac386cff3fe447c","observation_id":"a01fdfc8-680c-41b3-bd5d-22b4f8366ac9","resolution":{"observed_at":"2026-05-20T21:23:44.400235Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.19950","last_updated":"2026-05-19T15:05:00Z","snapshot_observed_at":"2026-07-06T23:30:35.042915Z","submitted_at":"2026-05-19T15:05:00Z","title":"AffectVerse: Emotional World Models for Multimodal Affective Computing","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T06:46:33.612905Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.19950"},"observation_digest":"sha256:f161aa67963fa15bc7628c6c2eeedfc68433824c30fdf85fd75374749e8f8ae6","observation_id":"c5e2dafd-2067-41b0-a964-7d8a0c76c25b","resolution":{"observed_at":"2026-05-20T06:48:05.774694Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2605.24024","last_updated":"2026-05-20T13:29:29Z","snapshot_observed_at":"2026-07-06T23:34:08.786038Z","submitted_at":"2026-05-20T13:29:29Z","title":"Mitigating Hallucinations in Large Vision-Language Models via Causal Route Gating","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-30T17:19:45.341282Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2605.24024"},"observation_digest":"sha256:3f6c2ecc90671ae550ac6270b1cdebdb59c6c06d9f15f72c279bc12350aa1b44","observation_id":"9c0c9dae-d5f3-4b24-a83f-7ccb7dd9480a","resolution":{"observed_at":"2026-06-30T17:24:56.634573Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.03793","last_updated":"2026-06-02T15:42:10Z","snapshot_observed_at":"2026-08-02T05:49:59.614618Z","submitted_at":"2026-06-02T15:42:10Z","title":"Exploring Adversarial Robustness and Safety Alignment in Multilingual Multi-Modal Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T10:10:38.261635Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.03793"},"observation_digest":"sha256:cf9e9f92c55d2b406172a5b7114a7efc00af128dbfc25212bee6e5cab973337f","observation_id":"7a0f665d-f0f7-49d3-9cc5-f925a4ff2880","resolution":{"observed_at":"2026-07-02T03:16:34.738320Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.11853","last_updated":"2026-06-10T09:30:25Z","snapshot_observed_at":"2026-08-02T20:07:37.622537Z","submitted_at":"2026-06-10T09:30:25Z","title":"Task-Aware Structured Memory for Dynamic Multi-modal In-Context Learning","version":1},"reference_index":123,"source":"arxiv_source","source_observed_at":"2026-06-27T10:28:11.440915Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.11853"},"observation_digest":"sha256:44941727dbc37fcbf697f0b6193af85e61f10e4c3637109b9cea547d1286511e","observation_id":"7547257a-0f44-45c3-8c3e-867ea7e2cf4a","resolution":{"observed_at":"2026-07-03T09:17:48.386689Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.12883","last_updated":"2026-06-11T04:19:32Z","snapshot_observed_at":"2026-08-02T22:02:12.932569Z","submitted_at":"2026-06-11T04:19:32Z","title":"The Hidden Power of Scaling Factor in LoRA Optimization","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-06-27T07:14:08.479610Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.12883"},"observation_digest":"sha256:11132ad75c17514bd83dc93649bedd5c4d49a00a3234ee39ab074d8aeb563bdc","observation_id":"ffa5f6a3-f059-4790-a24d-e68400af9c00","resolution":{"observed_at":"2026-07-03T14:08:21.974063Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.17030","last_updated":"2026-06-17T13:54:57Z","snapshot_observed_at":"2026-08-01T21:48:31.832288Z","submitted_at":"2026-06-15T17:52:31Z","title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-06-27T04:19:26.332718Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.17030"},"observation_digest":"sha256:c24048dea50aa9c374dd6195c2a764e0e7a7212c23d46032b39722866998c0ad","observation_id":"19d3cfb6-64a2-42d2-b9c7-9d2bd0607996","resolution":{"observed_at":"2026-07-03T17:18:43.806230Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.19706","last_updated":"2026-06-18T02:05:14Z","snapshot_observed_at":"2026-07-06T23:54:56.685241Z","submitted_at":"2026-06-18T02:05:14Z","title":"NEST: Narrative Event Structures in Time for Long Video Understanding","version":1},"reference_index":288,"source":"arxiv_source","source_observed_at":"2026-06-26T17:57:55.366051Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.19706"},"observation_digest":"sha256:8e76efa7521f0b03bcc2f7e4568484baee9a4a12ba305399b9999afc26e9fd5b","observation_id":"e7d40f5e-62f0-4007-ace4-21dc05b5f166","resolution":{"observed_at":"2026-07-04T03:29:31.187726Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-07-06T23:56:40.795364Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":150,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:e8a957622942537b18799481334d0dbef0209350d374f1571861b2c29c2818d6","observation_id":"18c50c91-7666-4f9d-8b9e-b539045cb3d6","resolution":{"observed_at":"2026-07-04T06:39:37.691004Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.24118","last_updated":"2026-06-23T04:09:05Z","snapshot_observed_at":"2026-08-03T04:47:08.013790Z","submitted_at":"2026-06-23T04:09:05Z","title":"An LMM for Precisely Grounding Elements in Documents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-26T01:51:50.043216Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.24118"},"observation_digest":"sha256:9d88e0500a042211f338a3655607b226935bd84a2b30b079c80a5ba40de95a1a","observation_id":"cb54ec4c-5023-406a-897a-55cc01b83f35","resolution":{"observed_at":"2026-07-04T15:09:54.720233Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.25325","last_updated":"2026-07-20T13:20:39Z","snapshot_observed_at":"2026-08-02T10:17:41.360200Z","submitted_at":"2026-06-24T02:43:26Z","title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-06-25T21:31:38.450382Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.25325"},"observation_digest":"sha256:b6c2db15c5b09483933833ac4f2ea97382c7dff0e98710a307daf0b7cc06240d","observation_id":"7e892e0e-68fe-4e70-862d-bb51d0142525","resolution":{"observed_at":"2026-07-04T19:20:06.429472Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-02T10:17:44.329479Z","title":"org/CorpusID:274859421","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.25325","last_updated":"2026-07-20T13:20:39Z","snapshot_observed_at":"2026-08-02T10:17:41.360200Z","submitted_at":"2026-06-24T02:43:26Z","title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-02T10:17:44.329479Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.25325"},"observation_digest":"sha256:b9279d03f876fe1488b2d2843039bb4bd41c490066951cf74b579e0161fc08bf","observation_id":"23fa663c-7f5f-495b-b8ff-27fcefa72f8e","resolution":{"observed_at":"2026-08-02T10:17:44.329479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.27161","last_updated":"2026-06-25T15:29:37Z","snapshot_observed_at":"2026-08-04T02:05:59.242279Z","submitted_at":"2026-06-25T15:29:37Z","title":"TOPS: First-Principles Visual Token Pruning via Constructing Token Optimal Preservation Sets for Efficient MLLM Inference","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-26T04:24:38.917137Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.27161"},"observation_digest":"sha256:0fd977df20656f0195d5368a4ae9e1ff63b1da55fbfa362f587c6cd937507e15","observation_id":"502db605-70bb-49ac-99d3-9aada034aebc","resolution":{"observed_at":"2026-07-04T14:09:53.838517Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.27652","last_updated":"2026-06-26T02:07:53Z","snapshot_observed_at":"2026-08-02T13:44:43.785281Z","submitted_at":"2026-06-26T02:07:53Z","title":"MER-R1: Multimodal Emotion Reasoning via Slow-Fast Thinking Synergy","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-29T05:06:09.216428Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.27652"},"observation_digest":"sha256:a160090e911514d696b5bb78d3e2ec3f8cb8a759f1bf4fd8ef7075ba2d089c4d","observation_id":"143b3b6d-b76f-414e-84ef-d5e764626aae","resolution":{"observed_at":"2026-06-29T18:23:51.443168Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2606.29847","last_updated":"2026-06-29T06:35:47Z","snapshot_observed_at":"2026-07-07T00:03:45.783349Z","submitted_at":"2026-06-29T06:35:47Z","title":"See Only When Needed: Context-Aware Attention Intervention for Mitigating Hallucinations in LVLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-30T06:47:01.711336Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2606.29847"},"observation_digest":"sha256:f4fea618ab1e6b20bbb3584e072f245244a433e255540bb9308436605c98698c","observation_id":"f55624f8-0cc2-48c3-89cd-6e9cd757d53e","resolution":{"observed_at":"2026-06-30T06:54:20.794831Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":"2304.14178","doi":"10.48550/arxiv.2304.14178","metadata_source":"pith","pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","venue":"cs.CL","work_id":"74a7deb6-48be-4132-9d35-882cc5870ebd","year":2023},"citing_paper":{"arxiv_id":"2607.02484","last_updated":"2026-07-02T17:50:57Z","snapshot_observed_at":"2026-07-07T00:07:56.573640Z","submitted_at":"2026-07-02T17:50:57Z","title":"Combating Textual Noise and Redundancy: Entropy-Aware Dense Visual Token Pruning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-07-03T14:47:35.377391Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2607.02484"},"observation_digest":"sha256:5c7f6f6bf97a41d501d283b9af8a436b696e59995a9297b4b16e43c70e33f79d","observation_id":"3b280dc7-d374-4a0c-a64f-c8c249e4708d","resolution":{"observed_at":"2026-07-03T14:48:32.415187Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-12T04:17:40.198357Z","title":"arXiv preprint arXiv:2304.14178 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03184","last_updated":"2026-07-03T10:42:34Z","snapshot_observed_at":"2026-08-03T20:34:31.286239Z","submitted_at":"2026-07-03T10:42:34Z","title":"BVS: Bayesian Visual Search with Multimodal Large Language Model for Fine-grained Perception","version":1},"reference_index":151,"source":"arxiv_source","source_observed_at":"2026-07-12T04:17:40.198357Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2607.03184"},"observation_digest":"sha256:db5f97ff23d58a55391f6b68ed5df479c28bb5fc6f55f4947ba1aba3deb85a86","observation_id":"5a096d5c-d394-4091-8171-91a6caf37252","resolution":{"observed_at":"2026-07-12T04:17:40.198357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-12T01:50:59.184754Z","title":"arXiv preprint arXiv:2304.14178 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03530","last_updated":"2026-07-03T17:59:58Z","snapshot_observed_at":"2026-08-04T18:44:44.436706Z","submitted_at":"2026-07-03T17:59:58Z","title":"MentalThink: Shaping Thoughts in Mental SVG World","version":1},"reference_index":206,"source":"arxiv_source","source_observed_at":"2026-07-12T01:50:59.184754Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2607.03530"},"observation_digest":"sha256:6dd2f63fa6d3c0d9bd3a67d551a93d3586c292b3a9aa28efe8f6a1502ed90216","observation_id":"78fb47b7-a639-4a07-93fc-ec7dfdae28ce","resolution":{"observed_at":"2026-07-12T01:50:59.184754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-07-14T03:31:19.309532Z","title":"arXiv:2304.14178 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11738","last_updated":"2026-07-13T16:00:03Z","snapshot_observed_at":"2026-08-02T08:52:21.513313Z","submitted_at":"2026-07-13T16:00:03Z","title":"Qwen-Audio-VAE Technical Report","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-07-14T03:31:19.309532Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2607.11738"},"observation_digest":"sha256:ade6908fbcdc4393ac3f698a49fce574e0e2c33a1dc0e99b350906164c33ce99","observation_id":"c2d6a28b-f309-4d26-ae2c-d32521d46a3a","resolution":{"observed_at":"2026-07-14T03:31:19.309532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-02T01:58:01.554825Z","title":"mplug-owl: Modularization empowers large language models with multimodality.arXiv preprint arXiv:2304.14178, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14499","last_updated":"2026-07-16T02:25:03Z","snapshot_observed_at":"2026-08-04T11:10:46.858246Z","submitted_at":"2026-07-16T02:25:03Z","title":"Contextualized Evaluation of Vision Language Models through Dynamic, Multi-turn Interactions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T01:58:01.554825Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2607.14499"},"observation_digest":"sha256:8cae8d6f541a1ab8e8465b41face06b8804e1e1a978decb2956b6e0208e32cf4","observation_id":"141a615b-57f2-4749-a930-356f8a16468b","resolution":{"observed_at":"2026-08-02T01:58:01.554825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2304.14178/citation-record","integrity":"/paper/2304.14178/integrity","json":"/paper/2304.14178/citation-record.json","paper":"/paper/2304.14178"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2204.14198","last_updated":"2022-11-15T23:07:37Z","snapshot_observed_at":"2026-07-06T13:05:12.350238Z","submitted_at":"2022-04-29T16:29:01Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","version":2},"cited_work":{"arxiv_id":"2204.14198","doi":"10.48550/arxiv.2204.14198","metadata_source":"pith","pith_arxiv_id":"2204.14198","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","venue":"cs.CV","work_id":"a110f764-38dc-41b2-a802-53744ecea1fc","year":2022},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2204.14198","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:9b743bf7a219596a63faa1e67650c2d1504531d952f7fa134eae7b65c22dddc0","observation_id":"c966cff0-1189-4175-b608-21e0d3cb2f22","resolution":{"observed_at":"2026-05-24T09:04:15.076622Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1504.00325","last_updated":"2015-04-03T20:21:16Z","snapshot_observed_at":"2026-08-04T18:05:27.145522Z","submitted_at":"2015-04-01T18:13:43Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","version":2},"cited_work":{"arxiv_id":"1504.00325","doi":"10.48550/arxiv.1504.00325","metadata_source":"pith","pith_arxiv_id":"1504.00325","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","venue":"cs.CV","work_id":"b3d6fb46-4169-4a28-8f7e-2ca6774211da","year":2015},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/1504.00325","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:17d2fef1a5544ddac4c61bff4fbc3a77f47235bd2b65e86cafb682068c2dd677","observation_id":"8ec80962-ecf4-4b0d-ab78-4a128e7c5401","resolution":{"observed_at":"2026-05-24T09:04:15.032403Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.02311","last_updated":"2022-10-05T06:02:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-05T16:11:45Z","title":"PaLM: Scaling Language Modeling with Pathways","version":5},"cited_work":{"arxiv_id":"2204.02311","doi":"10.48550/arxiv.2204.02311","metadata_source":"pith","pith_arxiv_id":"2204.02311","snapshot_observed_at":"2026-07-10T12:07:03.432645Z","title":"PaLM: Scaling Language Modeling with Pathways","venue":"cs.CL","work_id":"a94f3ef7-2c49-4445-93fe-6ec16aafd966","year":2022},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2204.02311","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:b7a61db5f076e1525160e6314e81322946239e9c3677679114ef2445ff36dc2d","observation_id":"ce41ffec-77dc-4295-9b45-32ed5be9ae1b","resolution":{"observed_at":"2026-05-24T09:04:15.080433Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-21T12:25:21.411055+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T12:25:21.411055+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11416","last_updated":"2022-12-06T21:39:48Z","snapshot_observed_at":"2026-07-06T14:08:18.855958Z","submitted_at":"2022-10-20T16:58:32Z","title":"Scaling Instruction-Finetuned Language Models","version":5},"cited_work":{"arxiv_id":"2210.11416","doi":"10.48550/arxiv.2210.11416","metadata_source":"pith","pith_arxiv_id":"2210.11416","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Scaling Instruction-Finetuned Language Models","venue":"cs.LG","work_id":"8405abb1-7558-4fdf-af24-f4c52fa77a06","year":2022},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2210.11416","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:18b81d2c8c9e53f9dc6b52e01432a99385adec57731afc8a278047e3cc85284c","observation_id":"31c7845c-ccb1-462c-a422-7c12891b1e07","resolution":{"observed_at":"2026-05-24T09:04:15.083930Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03378","last_updated":"2023-03-06T18:58:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-06T18:58:06Z","title":"PaLM-E: An Embodied Multimodal Language Model","version":1},"cited_work":{"arxiv_id":"2303.03378","doi":"10.48550/arxiv.2303.03378","metadata_source":"pith","pith_arxiv_id":"2303.03378","snapshot_observed_at":"2026-07-11T00:27:51.904332Z","title":"PaLM-E: An Embodied Multimodal Language Model","venue":"cs.LG","work_id":"5b99811a-1d93-47e2-9d59-f4045a0b74a2","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2303.03378","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:d25e64254deed4e90304daf670053e4040500c5420a8db6face21c7758a3b285","observation_id":"9e89c9d5-617c-40b2-8dd6-4055bd7caac9","resolution":{"observed_at":"2026-05-24T09:04:15.068093Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-09T19:19:06.044727+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T19:19:06.044727+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":"2301.12597","doi":"10.48550/arxiv.2301.12597","metadata_source":"pith","pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","venue":"cs.CV","work_id":"63d03f4d-15f4-4583-8286-913c19f02294","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:00cc60c2cb219d9770b765d841b755260f464d4f79442c0bd913a494b1fbbc73","observation_id":"1a325725-06b9-4be4-b2ca-1766b6c1bc84","resolution":{"observed_at":"2026-05-24T09:04:15.087475Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08485","last_updated":"2023-12-11T17:46:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-17T17:59:25Z","title":"Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2304.08485","doi":"10.48550/arxiv.2304.08485","metadata_source":"pith","pith_arxiv_id":"2304.08485","snapshot_observed_at":"2026-07-10T11:17:03.636249Z","title":"Visual Instruction Tuning","venue":"cs.CV","work_id":"68be622d-a6dc-4a13-82de-e3054a3dc509","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2304.08485","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:b7f84856f43fd68f26242f6c21da2717ff4e6333afca29cadcf58522dcf5bbdd","observation_id":"2e18a3f9-186c-4bc8-b2f8-cd0cf5a4288c","resolution":{"observed_at":"2026-05-24T09:04:15.028188Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:44.742124+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:44.742124+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:5b42bf2453d853709e65e3c36f079af081306ff1ea5f1226e4facdca4ee718a5","observation_id":"350f9dfe-3a83-4c0d-a890-cb095211731b","resolution":{"observed_at":"2026-05-24T09:04:15.024208Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":"2203.02155","doi":"10.1007/s00354-022-00198-8","metadata_source":"pith","pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training language models to follow instructions with human feedback","venue":"cs.CL","work_id":"52aff42f-4fa9-4fcf-bdb3-1459b9bebf65","year":2022},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:e54e0b26135414d53c3528c40e12ca109e8a08167c9cf3fa3d59a666615f8c54","observation_id":"64cc3e3f-973d-405a-a234-24816a57a3b4","resolution":{"observed_at":"2026-05-24T09:04:15.048146Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.05100","last_updated":"2023-06-27T09:57:58Z","snapshot_observed_at":"2026-08-04T18:56:03.233715Z","submitted_at":"2022-11-09T18:48:09Z","title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","version":4},"cited_work":{"arxiv_id":"2211.05100","doi":"10.48550/arxiv.2211.05100","metadata_source":"pith","pith_arxiv_id":"2211.05100","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","venue":"cs.CL","work_id":"337ba690-f35d-4154-9450-8edf4bc9f488","year":2022},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2211.05100","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:ee299708ac23a05631b188aebc77b7389afbd5f754805cc5e26da6be9626d719","observation_id":"8db6f3b4-7956-4623-a776-c4e649a9554d","resolution":{"observed_at":"2026-05-24T09:04:15.040147Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-25T16:25:57.975024+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T16:25:57.975024+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02114","last_updated":"2021-11-03T10:16:39Z","snapshot_observed_at":"2026-08-02T08:12:49.547570Z","submitted_at":"2021-11-03T10:16:39Z","title":"LAION-400M: Open Dataset of CLIP-Filtered 400 Million Image-Text Pairs","version":1},"cited_work":{"arxiv_id":"2111.02114","doi":"10.48550/arxiv.2111.02114","metadata_source":"pith","pith_arxiv_id":"2111.02114","snapshot_observed_at":"2026-07-10T15:47:23.233430Z","title":"LAION-400M: Open Dataset of CLIP-Filtered 400 Million Image-Text Pairs","venue":"cs.CV","work_id":"fbf16034-512e-4dd9-a0bd-7ddd23f532a6","year":2021},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2111.02114","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:fa39a75a59f44e88db04a89f679fed08e93fcce5266954a44677c567ea9a4264","observation_id":"c25a1df7-1ab1-4632-80d6-320c4f4b13db","resolution":{"observed_at":"2026-05-24T09:04:15.036222Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:40.426779+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:40.426779+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17580","last_updated":"2023-12-03T18:17:21Z","snapshot_observed_at":"2026-08-03T00:52:54.308486Z","submitted_at":"2023-03-30T17:48:28Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","version":4},"cited_work":{"arxiv_id":"2303.17580","doi":"10.48550/arxiv.2303.17580","metadata_source":"pith","pith_arxiv_id":"2303.17580","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","venue":"cs.CL","work_id":"f20ed1da-2676-4598-a11b-54549718735b","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2303.17580","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:49459736f6a48b335ffdc8c3bf8803334b7c252dbd69080249c35ed6929e0983","observation_id":"bdf32398-5d4c-470f-a001-a008a26c0be8","resolution":{"observed_at":"2026-05-24T09:04:15.044052Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:34.235168+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:34.235168+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-07-11T03:47:48.508181Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:31ceb7f9f99baa5b95134f8d1648c49c7e64627eb9b8ae21ad4ff95a59239886","observation_id":"f7883de6-9577-4d88-a56c-ef3b9b8c50b3","resolution":{"observed_at":"2026-05-24T09:04:15.097534Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10560","last_updated":"2023-05-25T23:50:07Z","snapshot_observed_at":"2026-07-06T14:33:11.945106Z","submitted_at":"2022-12-20T18:59:19Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","version":2},"cited_work":{"arxiv_id":"2212.10560","doi":"10.1145/3209978.3210080","metadata_source":"pith","pith_arxiv_id":"2212.10560","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","venue":"cs.CL","work_id":"d0018767-775d-406e-861d-539ed681ff73","year":2022},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2212.10560","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:bd6df469f39be30a157d29d9a4daeb45bdb51f6b3f34fb790cc95efa680413f6","observation_id":"477faf47-d840-4a0b-a645-1d5be12e8edf","resolution":{"observed_at":"2026-05-24T09:04:15.064182Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10560","last_updated":"2023-05-25T23:50:07Z","snapshot_observed_at":"2026-07-06T14:33:11.945106Z","submitted_at":"2022-12-20T18:59:19Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","version":2},"cited_work":{"arxiv_id":"2212.10560","doi":"10.1145/3209978.3210080","metadata_source":"pith","pith_arxiv_id":"2212.10560","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","venue":"cs.CL","work_id":"d0018767-775d-406e-861d-539ed681ff73","year":2022},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2212.10560","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:f21b41c08a14133e898bc003b97797366b24338607e73bcab311ebdfc03122bf","observation_id":"3ddb50c2-298c-4961-b890-67743b597329","resolution":{"observed_at":"2026-05-24T09:04:14.794976Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.01196","last_updated":"2023-12-02T21:05:22Z","snapshot_observed_at":"2026-07-06T15:11:35.938311Z","submitted_at":"2023-04-03T17:59:09Z","title":"Baize: An Open-Source Chat Model with Parameter-Efficient Tuning on Self-Chat Data","version":4},"cited_work":{"arxiv_id":"2304.01196","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.01196","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Baize: An open-source chat model with parameter-efficient tuning on self-chat data","venue":null,"work_id":"687c8f3b-3ae4-46c3-ac7d-322ba05df93f","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2304.01196","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:a36ac8617b3f6dfce2723f315024e3e1a9b3dd522a4f393739016cf7d6cfb251","observation_id":"e2490bfd-d056-4063-8440-6ec0db0165a3","resolution":{"observed_at":"2026-05-24T09:04:15.052649Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.00402","last_updated":"2023-02-01T12:40:03Z","snapshot_observed_at":"2026-07-06T14:47:00.945435Z","submitted_at":"2023-02-01T12:40:03Z","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","version":1},"cited_work":{"arxiv_id":"2302.00402","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2302.00402","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"mplug-2: A modularized multi-modal foundation model across text, image and video","venue":null,"work_id":"e48f6577-0181-4115-9570-ca671541d2a8","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2302.00402","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:6207b502cc1cc05c9bc9722255523e7e5c9425825dd5242489bf1a0d7bc689ff","observation_id":"71a320d5-2ef4-4aa6-a654-0a2a5620c7ba","resolution":{"observed_at":"2026-05-24T09:04:15.057623Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.14546","last_updated":"2022-12-30T04:27:01Z","snapshot_observed_at":"2026-07-06T14:36:01.830787Z","submitted_at":"2022-12-30T04:27:01Z","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","version":1},"cited_work":{"arxiv_id":"2212.14546","doi":"10.48550/arxiv.2212.14546","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.14546","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URL https://doi.org/10.48550/arXiv.2212.14546","venue":null,"work_id":"a2801809-89ae-4e45-9411-f3814d6ea3a3","year":2021},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2212.14546","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:7c7e477412571a2dde0c9a973cfaf358ca8109997f9f63d1a681045d88241d57","observation_id":"10d8118a-1f5e-44b4-ae73-3d4294e0bd8c","resolution":{"observed_at":"2026-05-24T09:04:15.071854Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.14546","last_updated":"2022-12-30T04:27:01Z","snapshot_observed_at":"2026-07-06T14:36:01.830787Z","submitted_at":"2022-12-30T04:27:01Z","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","version":1},"cited_work":{"arxiv_id":"2212.14546","doi":"10.48550/arxiv.2212.14546","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.14546","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"URL https://doi.org/10.48550/arXiv.2212.14546","venue":null,"work_id":"a2801809-89ae-4e45-9411-f3814d6ea3a3","year":2021},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2212.14546","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:f2ccd6fe8d89599280d0083c9234c87e9906e0cce4a6dc599001147f3cc238e7","observation_id":"28c91947-cc6a-49f5-8f2a-b9382edf2624","resolution":{"observed_at":"2026-05-24T09:04:14.790817Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06939","last_updated":"2023-10-28T04:19:41Z","snapshot_observed_at":"2026-07-06T15:15:39.104303Z","submitted_at":"2023-04-14T06:17:46Z","title":"Multimodal C4: An Open, Billion-scale Corpus of Images Interleaved with Text","version":3},"cited_work":{"arxiv_id":"2304.06939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.06939","snapshot_observed_at":"2026-07-01T22:26:17.900939Z","title":"Multimodal c4: An open, billion-scale corpus of images interleaved with text","venue":null,"work_id":"fc67e4a1-354a-4784-ac5b-f4d2f4f815d4","year":2023},"citing_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-24T09:02:31.211260Z"},"links":{"cited_paper":"/paper/2304.06939","citing_paper":"/paper/2304.14178"},"observation_digest":"sha256:7219160562fbc03f6c5dfb8937b92d68eca4cde505a76614ba308cf908d94cfa","observation_id":"e1a8c733-b4fb-417e-8c6c-509b6f930665","resolution":{"observed_at":"2026-05-24T09:04:15.092474Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality"},"reference_resolution":{"displayed":20,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":19,"verified_fuzzy":0},"total_outbound_references":20},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 20 of 20 outbound references and 100 inbound Pith citation observations for arXiv:2304.14178."}