{"as_of":"2026-08-23T10:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:adc681bc993c4eed7928d5fe696058c7719cd921fc76fbf4a6d53c590f9bd074","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":42,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:19:37.086555Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:48:56.198380Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-08-20T02:58:52.302783Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-17T00:05:03.547664Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2406.16860"},"observation_digest":"sha256:d7da99e458a4687af1c4ace000f712aa1c969647a1f05d6ccc386c32ffc4d0ac","observation_id":"3a50e86b-4986-4c4a-b5cd-c8852fac54da","resolution":{"observed_at":"2026-05-17T00:05:03.916736Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-08-21T07:48:51.700518Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T21:07:31.387726Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2408.01800"},"observation_digest":"sha256:543b2f5c0bbdca21007134ae365e310b9ac4279f6fbd28ddfa07848866c197e4","observation_id":"60f2322e-4ed9-4fe8-ad74-0d9acc528876","resolution":{"observed_at":"2026-05-10T21:07:31.947656Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-08-14T12:07:36.260067Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:a00caf5fdbc246f72aba6722b2896453825a27d3c46b855a6dedb11fbed735df","observation_id":"d7c5483a-670c-4b63-bdef-68f7bf107a24","resolution":{"observed_at":"2026-05-23T19:43:23.783118Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2410.13848","last_updated":"2024-10-17T17:58:37Z","snapshot_observed_at":"2026-08-03T03:16:45.693966Z","submitted_at":"2024-10-17T17:58:37Z","title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-15T22:09:16.001309Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2410.13848"},"observation_digest":"sha256:ad28f0e760840561d5416a751e28109640104b63f4080ae5521ae39146f707fb","observation_id":"27fa901c-1d2b-47ab-b17f-7e3e7b658d12","resolution":{"observed_at":"2026-05-15T22:09:16.148867Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-12T19:33:00.770072Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.10640","last_updated":"2024-11-16T00:14:51Z","snapshot_observed_at":"2026-08-17T21:54:44.695295Z","submitted_at":"2024-11-16T00:14:51Z","title":"BlueLM-V-3B: Algorithm and System Co-Design for Multimodal Large Language Models on Mobile Devices","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T19:33:00.770072Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2411.10640"},"observation_digest":"sha256:5e69b36111f73e2dece1a442253b468fcadd93520af61afad6024a96e17ec150","observation_id":"763d7f88-6794-437d-8ca5-263415c11c38","resolution":{"observed_at":"2026-08-12T19:33:00.770072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-11T23:36:24.718590Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02368","last_updated":"2024-12-03T10:52:06Z","snapshot_observed_at":"2026-08-19T22:51:25.169366Z","submitted_at":"2024-12-03T10:52:06Z","title":"ScImage: How Good Are Multimodal Large Language Models at Scientific Text-to-Image Generation?","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T23:36:24.718590Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.02368"},"observation_digest":"sha256:5fd09b5f35582129c3292a7e0bfe702f14b2932601c53cb2bd7ba8d977103bf9","observation_id":"601ddb87-e901-45fb-aa69-054dc81ebdb7","resolution":{"observed_at":"2026-08-11T23:36:24.718590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-11T21:29:59.953617Z","title":"Multimodal arxiv: A dataset for improving scien- tific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.04447","last_updated":"2025-04-11T07:10:02Z","snapshot_observed_at":"2026-08-14T15:38:54.058153Z","submitted_at":"2024-12-05T18:57:23Z","title":"EgoPlan-Bench2: A Benchmark for Multimodal Large Language Model Planning in Real-World Scenarios","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T21:29:59.953617Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.04447"},"observation_digest":"sha256:fabed7fb75551c0929b8a4005d18a5283427c6917a4e5f47c67818f7796c384c","observation_id":"d8d921e0-e543-4d53-8b04-811b89fa6e3d","resolution":{"observed_at":"2026-08-11T21:29:59.953617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":132,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:578aa8375d2054ed6d957bfba12ea4c78a5bf1156a42d1eaaa945a49d053c8a4","observation_id":"53b14f80-db4f-486f-b7ba-a97b6c857c51","resolution":{"observed_at":"2026-05-10T13:23:58.133266Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-11T20:13:47.564813Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.05983","last_updated":"2025-07-26T14:45:01Z","snapshot_observed_at":"2026-08-14T09:40:33.773143Z","submitted_at":"2024-12-08T16:10:42Z","title":"Chimera: Improving Generalist Model with Domain-Specific Experts","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T20:13:47.564813Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.05983"},"observation_digest":"sha256:990072b25597444510f3a2008c4edb420fdda679e53bd5bb8dfb807cdf2cec12","observation_id":"f7fd042b-5362-4d7c-b127-ea1c40380f22","resolution":{"observed_at":"2026-08-11T20:13:47.564813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-11T17:02:16.804403Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.09530","last_updated":"2024-12-12T18:20:41Z","snapshot_observed_at":"2026-08-15T11:11:08.377686Z","submitted_at":"2024-12-12T18:20:41Z","title":"Dynamic-VLM: Simple Dynamic Visual Token Compression for VideoLLM","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T17:02:16.804403Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.09530"},"observation_digest":"sha256:d4a504206a3fda7831b9931863fe134cc89d74cd913ad7212a727eaa87920f39","observation_id":"8c7cbd6e-9714-4531-9133-d38be1457413","resolution":{"observed_at":"2026-08-11T17:02:16.804403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2412.10302","last_updated":"2024-12-13T17:37:48Z","snapshot_observed_at":"2026-08-07T03:01:29.031129Z","submitted_at":"2024-12-13T17:37:48Z","title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-11T10:09:21.542356Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.10302"},"observation_digest":"sha256:6665db2c0ed24aeac000f27aef49bb0e628d16f1645fd3225a257381c0d95344","observation_id":"f70a4bce-d6c9-4da7-acce-b9facd44ab2f","resolution":{"observed_at":"2026-05-11T10:09:23.926362Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-11T18:19:03.880670Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12150","last_updated":"2024-12-11T05:29:54Z","snapshot_observed_at":"2026-08-14T08:56:00.251000Z","submitted_at":"2024-12-11T05:29:54Z","title":"Rethinking Comprehensive Benchmark for Chart Understanding: A Perspective from Scientific Literature","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T18:19:03.880670Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.12150"},"observation_digest":"sha256:fd9048039d4a4d07b8ee286a06399c19da91170209a9c1d845582671892812f1","observation_id":"9527b3a2-c3fc-4e2a-8f70-92a477a19b5b","resolution":{"observed_at":"2026-08-11T18:19:03.880670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2412.14164","last_updated":"2024-12-18T18:58:50Z","snapshot_observed_at":"2026-08-17T07:24:03.168446Z","submitted_at":"2024-12-18T18:58:50Z","title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","version":1},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-05-17T07:51:12.953777Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2412.14164"},"observation_digest":"sha256:b807357da1394e908a0c5e4306c446ac41cb55cd29997b5bef4ccbaaf971ef79","observation_id":"c93a2ead-e820-4d5d-94bf-0dac3f7f4884","resolution":{"observed_at":"2026-05-17T07:51:13.282378Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-10T18:04:34.374236Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.14818","last_updated":"2025-01-20T18:40:47Z","snapshot_observed_at":"2026-08-11T01:59:38.596539Z","submitted_at":"2025-01-20T18:40:47Z","title":"Eagle 2: Building Post-Training Data Strategies from Scratch for Frontier Vision-Language Models","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-10T18:04:34.374236Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2501.14818"},"observation_digest":"sha256:f13a3e6e13d76427109610c3e7f7ad9b166c76c55376411444460a50dc4d50f1","observation_id":"d046f113-cf1f-414c-9e9d-f21215c262d6","resolution":{"observed_at":"2026-08-10T18:04:34.374236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2502.02871","last_updated":"2026-04-20T02:18:01Z","snapshot_observed_at":"2026-08-17T11:21:07.508926Z","submitted_at":"2025-02-05T04:05:27Z","title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-23T04:30:38.804702Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2502.02871"},"observation_digest":"sha256:7107e42727ab80fe130e2b9cac4b491291e002d33f1e667ddf4873800f3b3dff","observation_id":"d238fc5d-c011-4afe-acad-2d0c4eec1d98","resolution":{"observed_at":"2026-05-23T04:32:33.338732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2504.09925","last_updated":"2026-04-29T06:12:36Z","snapshot_observed_at":"2026-08-16T10:58:08.667292Z","submitted_at":"2025-04-14T06:33:29Z","title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-22T19:49:00.961388Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2504.09925"},"observation_digest":"sha256:7f8b53f2f74233653b5a60537d2c4a8442472444415db6b27ed08b126bfbcb6d","observation_id":"238d0a45-84fd-4159-8213-2acd3e8df778","resolution":{"observed_at":"2026-05-22T19:52:01.925419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-16T12:19:37.086555Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.13180","last_updated":"2025-07-23T19:22:35Z","snapshot_observed_at":"2026-08-18T03:09:47.854491Z","submitted_at":"2025-04-17T17:59:56Z","title":"PerceptionLM: Open-Access Data and Models for Detailed Visual Understanding","version":3},"reference_index":134,"source":"pdf_text","source_observed_at":"2026-08-16T12:19:37.086555Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2504.13180"},"observation_digest":"sha256:1fb93de98a5b33127407d7a937fd2d727d8133297b0f9b2219f9fbc1af58eb86","observation_id":"26d54d24-5ff5-4297-9863-0ca74008e9f4","resolution":{"observed_at":"2026-08-16T12:19:37.086555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-15T20:17:23.409588Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models.arXiv preprint arXiv:2403.00231, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13427","last_updated":"2025-06-05T05:29:41Z","snapshot_observed_at":"2026-08-16T02:17:25.109640Z","submitted_at":"2025-05-19T17:55:08Z","title":"MM-PRM: Enhancing Multimodal Mathematical Reasoning with Scalable Step-Level Supervision","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T20:17:23.409588Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2505.13427"},"observation_digest":"sha256:1bb3cc19b3233fae8b9f7a71e588c0d7bbe2d91b4b969e0b864aab4dfe4be3a0","observation_id":"ebfac55c-0657-459d-915c-6b25fd72339e","resolution":{"observed_at":"2026-08-15T20:17:23.409588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-07T12:54:18.556271Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23242","last_updated":"2025-05-29T08:46:03Z","snapshot_observed_at":"2026-08-12T23:37:03.930593Z","submitted_at":"2025-05-29T08:46:03Z","title":"ChartMind: A Comprehensive Benchmark for Complex Real-world Multimodal Chart Question Answering","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T12:54:18.556271Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2505.23242"},"observation_digest":"sha256:2c20508a5032f5ea6f085005e2b78aa0eb95918238017f2666590be1a93d31fe","observation_id":"24f3f26e-458f-420b-9245-9692e789d51f","resolution":{"observed_at":"2026-08-07T12:54:18.556271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-06T17:31:05.650761Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10787","last_updated":"2025-07-14T20:35:25Z","snapshot_observed_at":"2026-08-21T05:52:29.824375Z","submitted_at":"2025-07-14T20:35:25Z","title":"Can Multimodal Foundation Models Understand Schematic Diagrams? An Empirical Study on Information-Seeking QA over Scientific Papers","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T17:31:05.650761Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2507.10787"},"observation_digest":"sha256:bae4a784c3627e13d299e043176ec4d9dc78fcdc07c3a60bf16c21dc738eaf2c","observation_id":"c8a5e407-34d7-4780-8eaf-88fd4a99f7d5","resolution":{"observed_at":"2026-08-06T17:31:05.650761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-06T15:52:14.564955Z","title":"arXiv preprint arXiv:2403.00231","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.14823","last_updated":"2025-07-20T05:00:42Z","snapshot_observed_at":"2026-08-13T06:52:56.828892Z","submitted_at":"2025-07-20T05:00:42Z","title":"FinChart-Bench: Benchmarking Financial Chart Comprehension in Vision-Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T15:52:14.564955Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2507.14823"},"observation_digest":"sha256:77f5c12704f455dd913ba5dedccebc3438f0d8e4e0b813faa1131ac423c1ee63","observation_id":"2181e933-b798-4bb5-957a-695d7eefdb49","resolution":{"observed_at":"2026-08-06T15:52:14.564955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-06T04:43:36.000015Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03164","last_updated":"2025-08-05T07:09:07Z","snapshot_observed_at":"2026-08-15T16:00:27.447250Z","submitted_at":"2025-08-05T07:09:07Z","title":"ChartCap: Mitigating Hallucination of Dense Chart Captioning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T04:43:36.000015Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2508.03164"},"observation_digest":"sha256:72a0546e44d8ffeef9d8118f6a11a78bd45c2bf9f07c58d6661f42972966d89e","observation_id":"e003bd98-94d9-497e-8c18-a6ab3d9cb463","resolution":{"observed_at":"2026-08-06T04:43:36.000015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-06T04:41:37.990911Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03175","last_updated":"2025-08-05T07:36:32Z","snapshot_observed_at":"2026-08-16T19:22:30.487110Z","submitted_at":"2025-08-05T07:36:32Z","title":"Adaptive Sparse Softmax: An Effective and Efficient Softmax Variant","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T04:41:37.990911Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2508.03175"},"observation_digest":"sha256:7a81d8378a66ffbacc548ea66ef904e1d4437f9f689a3f96899f973b8379e72c","observation_id":"871df512-c395-40d8-bb72-368936f975b7","resolution":{"observed_at":"2026-08-06T04:41:37.990911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-05T23:35:21.414530Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.05053","last_updated":"2025-08-07T06:10:15Z","snapshot_observed_at":"2026-08-19T17:09:34.159699Z","submitted_at":"2025-08-07T06:10:15Z","title":"Finding Needles in Images: Can Multimodal LLMs Locate Fine Details?","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T23:35:21.414530Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2508.05053"},"observation_digest":"sha256:3bdf236f2bf8179d9ddde3058dec72f7fde2142e2395427493e934b52a69efc5","observation_id":"5f894236-9082-4dc2-b385-e47f7738b3c1","resolution":{"observed_at":"2026-08-05T23:35:21.414530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-05T20:28:46.452098Z","title":"Li et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10955","last_updated":"2025-08-14T07:25:45Z","snapshot_observed_at":"2026-08-20T01:25:29.959449Z","submitted_at":"2025-08-14T07:25:45Z","title":"Empowering Multimodal LLMs with External Tools: A Comprehensive Survey","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T20:28:46.452098Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2508.10955"},"observation_digest":"sha256:22e28b2c5ce89b26ed1ee88b5b1e8f9b339f5d6e0c65caac95ad08f229d236e3","observation_id":"7da3b7e4-3912-42b7-8476-7b73c77741b9","resolution":{"observed_at":"2026-08-05T20:28:46.452098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-05T20:02:29.362654Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision- language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.15810","last_updated":"2025-08-15T08:41:33Z","snapshot_observed_at":"2026-08-13T17:36:54.054760Z","submitted_at":"2025-08-15T08:41:33Z","title":"Detecting Hope, Hate, and Emotion in Arabic Textual Speech and Multi-modal Memes Using Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T20:02:29.362654Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2508.15810"},"observation_digest":"sha256:d7bea2422fe9abe243ba951a472abb79907fb189fe9a2fd8afcc49c8959cd3c1","observation_id":"3eface75-9228-4622-b38e-64ada7d9de03","resolution":{"observed_at":"2026-08-05T20:02:29.362654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-04T19:25:50.051755Z","title":"InInternational conference on ma- chine learning, pages 19730–19742","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.09307","last_updated":"2025-09-11T09:50:16Z","snapshot_observed_at":"2026-08-18T18:49:23.652082Z","submitted_at":"2025-09-11T09:50:16Z","title":"Can Multimodal LLMs See Materials Clearly? A Multimodal Benchmark on Materials Characterization","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T19:25:50.051755Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2509.09307"},"observation_digest":"sha256:2dbeaed749a5fe046505993d2ecaaa67464409388b86c1bb9f893843f61820e6","observation_id":"54c5659c-cbc6-4c45-b550-49f93447698b","resolution":{"observed_at":"2026-08-04T19:25:50.051755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2511.05271","last_updated":"2026-03-11T08:46:41Z","snapshot_observed_at":"2026-07-06T22:35:13.699594Z","submitted_at":"2025-11-07T14:31:20Z","title":"DeepEyesV2: Toward Agentic Multimodal Model","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T05:32:29.266583Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2511.05271"},"observation_digest":"sha256:a290353e9f47708025c1122702fc71f93accc271c005ecfbc9f3ccad14ac1a1d","observation_id":"d6317614-23f2-40cf-9376-adde0a86f484","resolution":{"observed_at":"2026-05-16T05:32:29.342491Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2511.13415","last_updated":"2026-05-09T06:27:59Z","snapshot_observed_at":"2026-08-11T13:51:01.613197Z","submitted_at":"2025-11-17T14:28:41Z","title":"Attention Grounded Enhancement for Visual Document Retrieval","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T20:53:15.920563Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2511.13415"},"observation_digest":"sha256:04d09fa32f8b1666623b96550dae3297c75f0c86aab3b9d3ac10268a5ad6462d","observation_id":"b22e4d8c-5e39-4e47-852b-d0e060a42f9c","resolution":{"observed_at":"2026-05-17T20:55:15.235200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-03T06:04:26.757824Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision- language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.00722","last_updated":"2026-07-02T09:40:35Z","snapshot_observed_at":"2026-08-20T08:09:55.720321Z","submitted_at":"2026-01-31T13:27:02Z","title":"Spectral Imbalance Causes Forgetting in Low-Rank Continual Adaptation","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T06:04:26.757824Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2602.00722"},"observation_digest":"sha256:0023a397f4b8163f37bf08803e1b990a68161bff4e86ed751dcca64c20c658c8","observation_id":"4f52da81-0a9f-4f5a-9d0a-ed1ddf0e6d11","resolution":{"observed_at":"2026-08-03T06:04:26.757824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2604.04172","last_updated":"2026-04-05T16:30:21Z","snapshot_observed_at":"2026-08-08T10:40:02.076658Z","submitted_at":"2026-04-05T16:30:21Z","title":"GENFIG1: Visual Summaries of Scholarly Work as a Challenge for Vision-Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-13T16:45:17.022062Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2604.04172"},"observation_digest":"sha256:f6ec34039cfa463de5477c1e3c88431ce4af1e1d5d3d0e99a7656f2d6a7796de","observation_id":"5ab9584a-fef9-4d1f-ba7b-4957d8e3abe5","resolution":{"observed_at":"2026-05-13T16:48:03.177074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2604.08211","last_updated":"2026-04-09T13:11:01Z","snapshot_observed_at":"2026-08-13T12:01:56.589694Z","submitted_at":"2026-04-09T13:11:01Z","title":"SciFigDetect: A Benchmark for AI-Generated Scientific Figure Detection","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T17:39:49.848699Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2604.08211"},"observation_digest":"sha256:52ff8ad805f3a44d38624e9c76704f961efaa8831f90c637631fc09e4a02b4c5","observation_id":"a57c095b-5aa7-4e33-b881-eb084d0b059e","resolution":{"observed_at":"2026-05-11T06:25:57.771036Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2604.21304","last_updated":"2026-04-27T19:05:57Z","snapshot_observed_at":"2026-08-13T13:24:39.066917Z","submitted_at":"2026-04-23T05:42:39Z","title":"PaperMind: Benchmarking Agentic Reasoning and Critique over Scientific Papers in Multimodal LLMs","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-09T21:09:26.626179Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2604.21304"},"observation_digest":"sha256:1120158bb6edfe204d7b5d12f5f08d75e37b194261387f7ffde90319187b7746","observation_id":"2cf12307-a70e-41da-8af0-c36a875a6a96","resolution":{"observed_at":"2026-05-11T14:41:54.443885Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2605.02730","last_updated":"2026-05-04T15:31:11Z","snapshot_observed_at":"2026-08-02T14:48:33.210016Z","submitted_at":"2026-05-04T15:31:11Z","title":"Perceptual Flow Network for Visually Grounded Reasoning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-08T18:40:55.753827Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2605.02730"},"observation_digest":"sha256:ac4d88fb0f259601d4937a2c6116f8bee316ce7132b3f038371451398922546b","observation_id":"1142182e-cf02-4766-a44b-f99b9fedb515","resolution":{"observed_at":"2026-05-09T06:15:38.923328Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2605.14938","last_updated":"2026-05-14T15:13:24Z","snapshot_observed_at":"2026-08-21T21:28:41.309743Z","submitted_at":"2026-05-14T15:13:24Z","title":"Octopus: History-Free Gradient Orthogonalization for Continual Learning in Multimodal Large Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-30T20:56:38.324342Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2605.14938"},"observation_digest":"sha256:01a2c3565335e68d9ec2ca989955c56c38b3fb67b9919387fb35ef2aaa8e200a","observation_id":"a6ff8011-8189-494e-a938-9a5a1407433a","resolution":{"observed_at":"2026-07-01T14:35:46.686843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2605.19852","last_updated":"2026-06-03T11:11:28Z","snapshot_observed_at":"2026-08-13T09:24:20.626075Z","submitted_at":"2026-05-19T13:44:26Z","title":"Are Tools Always Beneficial? Learning to Invoke Tools Adaptively for Dual-Mode Multimodal LLM Reasoning","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-20T06:16:47.650748Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2605.19852"},"observation_digest":"sha256:c37a8808794c6121eea8f29b2f5d98894aa4b49a152e690b8aed6dd87d059fcd","observation_id":"056fba95-f8ea-4e27-99e0-371d26e3b8d1","resolution":{"observed_at":"2026-05-20T06:18:05.218966Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2605.20177","last_updated":"2026-05-19T17:58:40Z","snapshot_observed_at":"2026-08-14T07:37:07.675123Z","submitted_at":"2026-05-19T17:58:40Z","title":"From Seeing to Thinking: Decoupling Perception and Reasoning Improves Post-Training of Vision-Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T05:13:03.237427Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2605.20177"},"observation_digest":"sha256:b64205b1ba420ac7dfe7c9b0f5e1dfd9dcd3c21a5c473e909aa3afda6474ad41","observation_id":"c5eaa9fa-2347-4767-9243-568217237c48","resolution":{"observed_at":"2026-05-20T05:13:21.507568Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2606.11853","last_updated":"2026-06-10T09:30:25Z","snapshot_observed_at":"2026-08-02T20:07:37.622537Z","submitted_at":"2026-06-10T09:30:25Z","title":"Task-Aware Structured Memory for Dynamic Multi-modal In-Context Learning","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-06-27T10:28:11.440915Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2606.11853"},"observation_digest":"sha256:ceafeb421e44af84676830eb2b5a71801ec9e71685cae9bb31ba359ce38c45eb","observation_id":"8d972fdb-c888-4348-b7f9-aeb63c740a87","resolution":{"observed_at":"2026-07-03T09:07:48.431141Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2606.18216","last_updated":"2026-06-16T17:46:02Z","snapshot_observed_at":"2026-08-10T09:12:54.601851Z","submitted_at":"2026-06-16T17:46:02Z","title":"Zone of Proximal Policy Optimization: Teacher in Prompts, Not Gradients","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-06-27T01:08:52.981296Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2606.18216"},"observation_digest":"sha256:f92d276ffefe050763996837892eb0369830a7cb10f913e308224586488578fd","observation_id":"28741f6f-23d8-48ae-b58a-05a2601e5872","resolution":{"observed_at":"2026-07-03T20:48:56.199914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-12T04:17:40.198357Z","title":"arXiv preprint arXiv:2403.00231 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03184","last_updated":"2026-07-03T10:42:34Z","snapshot_observed_at":"2026-08-05T15:07:18.952856Z","submitted_at":"2026-07-03T10:42:34Z","title":"BVS: Bayesian Visual Search with Multimodal Large Language Model for Fine-grained Perception","version":1},"reference_index":168,"source":"arxiv_source","source_observed_at":"2026-07-12T04:17:40.198357Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2607.03184"},"observation_digest":"sha256:b0d95f87a1cb796de74ae31cf4b1e1adfddf394ddb7b4f359d7a61504a9e27ff","observation_id":"58ba945a-b848-45bf-bee5-0fa8910d1de3","resolution":{"observed_at":"2026-07-12T04:17:40.198357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-08T17:13:26.025156Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models.arXiv preprint arXiv:2403.00231, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05249","last_updated":"2026-08-07T01:16:18Z","snapshot_observed_at":"2026-08-16T09:44:03.621231Z","submitted_at":"2026-08-05T15:55:15Z","title":"PRISM: Priority-aware Rubric Internalization via Structured Multimodal Data Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T17:13:26.025156Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2608.05249"},"observation_digest":"sha256:57893fce3feeb4b6d2fbc8da3882e1b40b6959ece97fda892aac54fbd3dfd961","observation_id":"c4c2cf58-85e1-4a9c-9d72-cfc0dda00e0d","resolution":{"observed_at":"2026-08-08T17:13:26.025156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-08-10T04:31:25.731380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models.arXiv preprint arXiv:2403.00231, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05249","last_updated":"2026-08-07T01:16:18Z","snapshot_observed_at":"2026-08-16T09:44:03.621231Z","submitted_at":"2026-08-05T15:55:15Z","title":"PRISM: Priority-aware Rubric Internalization via Structured Multimodal Data Synthesis","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T04:31:25.731380Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2608.05249"},"observation_digest":"sha256:7159fd986498b82c5037be8f6bb5c374c3d934dcecf8b0551c6092a1977f97d9","observation_id":"03777384-0678-4999-9811-655003dc4633","resolution":{"observed_at":"2026-08-10T04:31:25.731380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.00231/citation-record","integrity":"/paper/2403.00231/integrity","json":"/paper/2403.00231/citation-record.json","paper":"/paper/2403.00231"},"outbound":[],"paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T14:13:54.531907Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 42 inbound Pith citation observations for arXiv:2403.00231."}