{"as_of":"2026-08-07T17:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9705d7bfdf8b4b4c361ec22aedb74d63e816a0850e4216180c43a413d224f3c8","coverage":[{"denominator":70,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":70,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T23:03:10.178730Z","state":"measured"},{"denominator":70,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":70,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.06009/citation-record","integrity":"/paper/2508.06009/integrity","json":"/paper/2508.06009/citation-record.json","paper":"/paper/2508.06009"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.131795Z","title":null,"venue":null,"work_id":"77f32fda-01a9-4d0c-ae52-f9fa37b7d1fe","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:05.479730Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:6a7fa422b46202b1c7dfc44d3c5a1b726912fc4e33957c0b919b561c888f0313","observation_id":"a3f53cd3-5ee1-46be-8659-9225c646233f","resolution":{"observed_at":"2026-08-05T23:03:11.136528Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.122722Z","title":null,"venue":null,"work_id":"b5eee6d5-379f-44cc-b3f6-89081d33b62b","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:05.548855Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:940972bc0e11862afd7f588bbdfcff9766da8c8d693f37bbb25109c85812f7d4","observation_id":"b38bca70-330c-4537-bf98-e86dfc74ca3c","resolution":{"observed_at":"2026-08-05T23:03:11.125949Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.113870Z","title":null,"venue":null,"work_id":"91e788c9-106b-4160-b263-260945ddea80","year":2019},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:05.630453Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:957773d0c00c59648bbc0a66a76fa06ee874e00eea8d11742ee31c3ec4af8bca","observation_id":"7e50cb86-3a88-40d2-bf01-b7e465fa5f7b","resolution":{"observed_at":"2026-08-05T23:03:11.116421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.104338Z","title":"D.; and Ammanabrolu, P","venue":null,"work_id":"abc70599-edb6-4685-8b4c-59b0ed1942da","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:05.698730Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:548c42dc555650fdce03e91acf86729e1f2cecff307df0c6f3bb3d2b7f7cc6ec","observation_id":"661b3ec9-63de-46cb-bba7-6d56bd1afe86","resolution":{"observed_at":"2026-08-05T23:03:11.107598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.095321Z","title":null,"venue":null,"work_id":"1598dfe8-a018-4f5b-9a76-c15569de83cd","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:05.782415Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:e370f1b213410f2d4b5b84ad7263956b55f9230ae689d67592b8f8749bc647c6","observation_id":"019de136-a5fb-49e7-b721-ea8225ea87f3","resolution":{"observed_at":"2026-08-05T23:03:11.098134Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.087421Z","title":"S.; Bahaj, S","venue":null,"work_id":"c7bf2da4-0de8-4088-a494-0b7922a4a642","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:05.902707Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:64579f0e38bff9dafe18003d316706a8974fbd45ca2318ce398e4b7f4096355b","observation_id":"67e0bbaa-98a7-4d03-be3f-2c61303d8811","resolution":{"observed_at":"2026-08-05T23:03:11.090252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-05T23:03:06.024471Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.024471Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:a97d646999799593df67b595847ae6c5ddcd11a877836d7a9acadb9a4c60cc55","observation_id":"67ae9123-8417-4077-898e-4bc46c7a5b94","resolution":{"observed_at":"2026-08-05T23:03:06.024471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T23:03:06.129793Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.129793Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:7db82aa9171e6681bda30ceafa5506981f2e98fe5c9e6e91b6e30a1ed8a65d2d","observation_id":"a0f2249e-6e3a-4c6a-9816-39f714cb71d7","resolution":{"observed_at":"2026-08-05T23:03:06.129793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.077259Z","title":null,"venue":null,"work_id":"bcfcd40f-568a-4783-8831-e242f25f3576","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.189698Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:ed1e80a41450b59e30bf8ab8b8ddb088aa04d4405db0d1d8c09228adfcc4a3e5","observation_id":"e144e3fe-8be4-40c8-a66c-c85a2f0d0488","resolution":{"observed_at":"2026-08-05T23:03:11.080301Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.063798Z","title":null,"venue":null,"work_id":"358ff2bf-13fb-42a1-a1ec-055ae3d10350","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.274113Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:482e2b8c4ddf6472fc304666552c948aeba1e6673b78038e682dd34a67abb923","observation_id":"7256ba6d-cb3d-432f-b1ec-11ae92db001e","resolution":{"observed_at":"2026-08-05T23:03:11.069911Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.050837Z","title":null,"venue":null,"work_id":"a5c009e0-24d2-4889-9ab0-35e5d7249a4e","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.405881Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:843ea71674e1e588b35e3f6aa187fde4970f6385c3c4774161bb71d72979a7f5","observation_id":"a97923c0-6d6c-47cb-a13c-450e89675f53","resolution":{"observed_at":"2026-08-05T23:03:11.054461Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.040499Z","title":null,"venue":null,"work_id":"7b52f7ef-e471-40eb-99ca-cd056c806cc5","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.513232Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:a26913277b94019987aec702cbc88f33c44a645f9ffce4b67240edf45a88769b","observation_id":"2f187234-bd6e-4fc0-917d-39b20e354fcd","resolution":{"observed_at":"2026-08-05T23:03:11.044836Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.030082Z","title":null,"venue":null,"work_id":"fbd8b174-7ec2-4c83-815c-93bcb89fd952","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.534871Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:0546fda8371f5c634ff4321b3f7f43951f7f31856991205e7f64b6448b3f3233","observation_id":"510143d2-1596-4e11-a7df-93bb2fa1d1c5","resolution":{"observed_at":"2026-08-05T23:03:11.033015Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11468","last_updated":"2025-04-10T16:54:05Z","snapshot_observed_at":"2026-08-05T02:57:13.928237Z","submitted_at":"2025-04-10T16:54:05Z","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.11468","snapshot_observed_at":"2026-08-05T23:03:06.609642Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.609642Z"},"links":{"cited_paper":"/paper/2504.11468","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:d1f5ddf76510e8f7f7a5340e18f0c9b9d4e821aac1bd184c4f22c6a6cc3faf11","observation_id":"52ab568b-7125-4c5f-8753-0f7549bbe07d","resolution":{"observed_at":"2026-08-05T23:03:06.609642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:06.826107Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.826107Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:a2f8b4865cef7575c26e09eaf9d2b909324597b47cdca99c6621b3b2a7056868","observation_id":"7463ee3b-7a8b-4b4c-8b17-39c278af18b2","resolution":{"observed_at":"2026-08-05T23:03:06.826107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-05T23:03:06.901216Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:06.901216Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:28a0c49bcbfe49a9510d523392c59f66f6709ab053d1e9b33d136f19405346c7","observation_id":"f661969b-fbb4-4be4-87a8-f55ab6236a67","resolution":{"observed_at":"2026-08-05T23:03:06.901216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-05T23:03:07.061721Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.061721Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:95d4cc1a2d1e05ae294f426cc76e6822a69e710ed56e7ea429235e7c08813c48","observation_id":"4bf6a102-a30a-416f-b249-14055def1dc4","resolution":{"observed_at":"2026-08-05T23:03:07.061721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17352","last_updated":"2025-11-11T08:13:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-21T17:52:43Z","title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.17352","snapshot_observed_at":"2026-08-05T23:03:07.169525Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.169525Z"},"links":{"cited_paper":"/paper/2503.17352","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:b4fd02728c3818fb663ffb579521d66c3e9142d06ed5ea11d4b97ffe06cd55d5","observation_id":"bb1a9588-635b-4ae0-9df9-a79c863ab9a3","resolution":{"observed_at":"2026-08-05T23:03:07.169525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:07.244354Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.244354Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:d4cb5a3de33443da5b40474f730e96e5676d9a0882908210f79f0857a1d248e5","observation_id":"e26bc2f0-7d3a-407e-8a53-524173d99e44","resolution":{"observed_at":"2026-08-05T23:03:07.244354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-07T17:13:10.057242Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00321","snapshot_observed_at":"2026-08-05T23:03:07.345505Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.345505Z"},"links":{"cited_paper":"/paper/2501.00321","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:08bdaf3085983c7d4d7464281f2d1e5789301fd28b6eb9ff626e839880a45d22","observation_id":"cb3702ed-2f72-4b02-b978-4fae2d648b4f","resolution":{"observed_at":"2026-08-05T23:03:07.345505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T23:03:07.437202Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.437202Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:16ff83a2784f7051331fbc9474f1a4cae84b067096eba8ab59679bd9b65da370","observation_id":"0c7f0ac6-c693-4f66-b267-b8e515ff00ee","resolution":{"observed_at":"2026-08-05T23:03:07.437202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14008","last_updated":"2024-06-06T13:19:44Z","snapshot_observed_at":"2026-08-03T03:39:09.398343Z","submitted_at":"2024-02-21T18:49:26Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14008","snapshot_observed_at":"2026-08-05T23:03:07.545378Z","title":"L.; Shen, J.; Hu, J.; Han, X.; Huang, Y.; Zhang, Y.; et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.545378Z"},"links":{"cited_paper":"/paper/2402.14008","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:6bc6c8fb0b39b80ca27446674aead97caf86c327dee84ccd03af900132246ae1","observation_id":"e0211619-0a7c-4cb0-a0f0-f8ff398862f6","resolution":{"observed_at":"2026-08-05T23:03:07.545378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T23:03:07.632051Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.632051Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:70f34c2a938fee177d6e2329b125578e2578679e6bdae90e5187d47fc72afa10","observation_id":"b1f05515-02b9-4a4c-9d86-6a0c6faa08c2","resolution":{"observed_at":"2026-08-05T23:03:07.632051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.11790","last_updated":"2026-06-23T12:36:04Z","snapshot_observed_at":"2026-07-06T20:23:36.052687Z","submitted_at":"2025-01-20T23:41:22Z","title":"Benchmarking LLMs' Mathematical Reasoning with Unseen Random Variables Questions","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.11790","snapshot_observed_at":"2026-08-05T23:03:07.734635Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.734635Z"},"links":{"cited_paper":"/paper/2501.11790","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:68f66a508e0c92856db52f5a7fb75cbbfd6c2b5b12fab47478adaca54d6b8992","observation_id":"899150c5-4673-40c3-b6a9-e19a38c1dc7a","resolution":{"observed_at":"2026-08-05T23:03:07.734635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17163","last_updated":"2026-05-26T01:50:12Z","snapshot_observed_at":"2026-08-07T14:52:59.748689Z","submitted_at":"2025-05-22T15:25:14Z","title":"OCR-Reasoning Benchmark: Unveiling the True Capabilities of MLLMs in Complex Text-Rich Image Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.17163","snapshot_observed_at":"2026-08-05T23:03:07.854784Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.854784Z"},"links":{"cited_paper":"/paper/2505.17163","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:99cb9b20a6320a19561832814cd6d0de515db180758e7f659a7efacc6e9ea4e3","observation_id":"429db48e-c95c-4a12-bdd3-388ffca76dd4","resolution":{"observed_at":"2026-08-05T23:03:07.854784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.020522Z","title":null,"venue":null,"work_id":"028dd2e4-8787-437a-a301-c51891a29797","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:07.949013Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:8fb7056f52be9027edb2091759c4d0c8b548c1791ee533da510c3055eb5ae95c","observation_id":"0fdc2059-8eac-43f0-8bff-108a0c209620","resolution":{"observed_at":"2026-08-05T23:03:11.023470Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00947","last_updated":"2025-07-13T22:29:38Z","snapshot_observed_at":"2026-08-03T17:23:10.606578Z","submitted_at":"2024-12-01T19:46:22Z","title":"VisOnlyQA: Large Vision Language Models Still Struggle with Visual Perception of Geometric Information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00947","snapshot_observed_at":"2026-08-05T23:03:08.012180Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.012180Z"},"links":{"cited_paper":"/paper/2412.00947","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:e5a29bc22485af8ea1d045ce4f533a628f4c21fa5a106f81168211125d79ca72","observation_id":"f4afe626-f042-4eb8-948c-6478f66f7fef","resolution":{"observed_at":"2026-08-05T23:03:08.012180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:11.010752Z","title":null,"venue":null,"work_id":"6b97ca65-fc5f-4e2b-a527-8a1ac808468e","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.096139Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:ab77e25da275b18933bc436697bbe1a6797d001e642f2790c788b11760e277fe","observation_id":"5f44f766-ee04-4484-9033-28be5d49da4d","resolution":{"observed_at":"2026-08-05T23:03:11.014370Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-06T04:28:38.934829Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-05T23:03:08.173192Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.173192Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:113a73c2417dfc7cb680bbbf846a0ce7e1b099023e86896f70eac4bdb340e71e","observation_id":"e446c60f-acc8-4f68-9a7d-af9aa250ad29","resolution":{"observed_at":"2026-08-05T23:03:08.173192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.06727","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.604731Z","title":null,"venue":null,"work_id":"a89d6f55-1c0b-4a72-af5b-bee0146a381d","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.264837Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:00529062e63a02c79d3f1ab0482a092d2344981df0f17e2ec75bdb59df6edc75","observation_id":"7458803d-b9af-4f7d-9139-0725b7e7c1df","resolution":{"observed_at":"2026-08-05T23:03:10.609040Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12023","last_updated":"2024-06-28T02:35:51Z","snapshot_observed_at":"2026-07-06T18:47:25.751118Z","submitted_at":"2024-06-28T02:35:51Z","title":"CMMaTH: A Chinese Multi-modal Math Skill Evaluation Benchmark for Foundation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12023","snapshot_observed_at":"2026-08-05T23:03:08.364386Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.364386Z"},"links":{"cited_paper":"/paper/2407.12023","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:987b73726995d2396db76a8c80f08e7aecbb771e4918b0bccd217c2bf2f7085d","observation_id":"3e9f344b-1c98-4941-af2c-1bf31b044908","resolution":{"observed_at":"2026-08-05T23:03:08.364386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-05T23:03:08.435992Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.435992Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:296fc7aaff057c47a82d3be4ec74128795fcb84b2e6f0d850f40063baffad0fd","observation_id":"65025616-6cc4-4899-8aaf-823dfe965e17","resolution":{"observed_at":"2026-08-05T23:03:08.435992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14295","last_updated":"2024-05-23T08:15:49Z","snapshot_observed_at":"2026-08-06T05:45:07.767677Z","submitted_at":"2024-05-23T08:15:49Z","title":"Focus Anywhere for Fine-grained Multi-page Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14295","snapshot_observed_at":"2026-08-05T23:03:08.515997Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.515997Z"},"links":{"cited_paper":"/paper/2405.14295","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:563a3ee68a1381acd8438b48e42027b5549951cc10e9fbda89c33835c33ae916","observation_id":"8066f43e-ca67-4b67-8ca5-d290acc7a358","resolution":{"observed_at":"2026-08-05T23:03:08.515997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.998766Z","title":null,"venue":null,"work_id":"d87d7364-8271-4da7-9223-3e2e089ccd94","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.606167Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:6edef135965813c8dcd318d5cc077306148e17796f88b3c55181cbee3cc122dc","observation_id":"380ac013-5ac1-43fe-b9ce-d221e9b4ed3c","resolution":{"observed_at":"2026-08-05T23:03:11.004621Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.989837Z","title":null,"venue":null,"work_id":"1e669355-c3ee-4877-9b16-a1077e3c588e","year":2023},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.696090Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:a0fc56827ea00ac8bc227afbaebf79b4f1d8e9efb518410d773d9838653d909a","observation_id":"5e9ad217-3b3c-4ffe-bc04-8c7437c425c5","resolution":{"observed_at":"2026-08-05T23:03:10.992808Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.10244","last_updated":"2022-03-19T05:00:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-19T05:00:30Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.10244","snapshot_observed_at":"2026-08-05T23:03:08.779737Z","title":"X.; Tan, J","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.779737Z"},"links":{"cited_paper":"/paper/2203.10244","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:d1d1a2947a8e642648e162757f05f7c962541157f156927cf2b3039a9af3cdda","observation_id":"accb2ee4-d4c1-4e0a-8015-82579f104c99","resolution":{"observed_at":"2026-08-05T23:03:08.779737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:08.881627Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:08.881627Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:ae33d29cc433328885a49c3dfd3b4d35dbc478efdd5e95113f6504f744ecbab5","observation_id":"45c6c99f-4dc9-45fe-87ad-1a2aef8d2741","resolution":{"observed_at":"2026-08-05T23:03:08.881627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:09.004368Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.004368Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:edbb8a33761d10e9599a0ee5ca0e8b6e0e5e969f62c26952a9773de2d3bbc09b","observation_id":"e2016163-bd89-441e-a1da-aa3a6e5fb986","resolution":{"observed_at":"2026-08-05T23:03:09.004368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:09.168731Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.168731Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:c5a8677a00ce6674dee57f229d708aec44ce517a5ca08e3d48a7ef0e3baf6e15","observation_id":"0f940b45-b68a-4cf9-acee-4d2492a24848","resolution":{"observed_at":"2026-08-05T23:03:09.168731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.965606Z","title":null,"venue":null,"work_id":"b0ec5d39-72b2-47b5-afc7-e79990f38c6c","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.300212Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:e44306975f0b55674a6a7d60205d9ef4c096188579dec09a5f12612e3f2253c5","observation_id":"c1d456e7-75c6-4dac-8b03-f18346f8f951","resolution":{"observed_at":"2026-08-05T23:03:10.968289Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.957987Z","title":null,"venue":null,"work_id":"0633b2ad-e01b-4e22-951f-ded56fcf52f7","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.418539Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:dd373e5d596c484010ad7e423c05f80ce5dfd7def5c2bde54e6ddee702668f98","observation_id":"907d46af-5286-4823-b493-aba88da5abf9","resolution":{"observed_at":"2026-08-05T23:03:10.960560Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.949987Z","title":null,"venue":null,"work_id":"d0b4f2bb-df77-4a74-aaaf-7dfcc4809b5d","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.585949Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:0e9d6fd3f4008c6a4a5a4f7f86904861c9ef0fbc3b905c8e711433a950e7cdbf","observation_id":"999c36b2-a78d-47d1-b25d-8364c7a3cbe8","resolution":{"observed_at":"2026-08-05T23:03:10.952770Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06167","last_updated":"2025-07-10T15:41:04Z","snapshot_observed_at":"2026-08-07T10:37:03.707334Z","submitted_at":"2025-07-08T16:47:16Z","title":"Skywork-R1V3 Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06167","snapshot_observed_at":"2026-08-05T23:03:09.720243Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.720243Z"},"links":{"cited_paper":"/paper/2507.06167","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:aefdb204d9a82822b0f7193706be8d1cb69a1dc45bcb8c3b48aa31241a8c956d","observation_id":"cbb68054-14f4-434d-bc7c-2dd937822bb7","resolution":{"observed_at":"2026-08-05T23:03:09.720243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.942062Z","title":"W.; Tay, Y.; Ruder, S.; Zhou, D.; et al","venue":null,"work_id":"661a44bc-ff23-4227-a895-63787c1ea7a0","year":2022},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.892367Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:b140567dc7b86a77d834e389a1a430adfb96879024423977a14bc89c8a884670","observation_id":"f226de3c-ad66-4d4e-854a-0c91a587c763","resolution":{"observed_at":"2026-08-05T23:03:10.944633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.934452Z","title":null,"venue":null,"work_id":"d0807617-b99a-469d-8681-ead233b32e61","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:09.990297Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:4faf4f8ddcb1749cbde518d92bd58068da1b1e7a84ab6e0ea2b36f26278ca0d1","observation_id":"13955807-f937-4d3e-adea-6aa6c22de0b5","resolution":{"observed_at":"2026-08-05T23:03:10.937022Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.926075Z","title":null,"venue":null,"work_id":"152d42cc-37f1-4a21-ad2f-f114a58fc463","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.101605Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:2df8084bc521e7eaefdb2456693a82036319a43858dae7bc12ef0d82367811fd","observation_id":"7d52edc9-2494-4423-99e0-35a83f14b4f1","resolution":{"observed_at":"2026-08-05T23:03:10.928498Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03569","last_updated":"2025-06-04T04:32:54Z","snapshot_observed_at":"2026-08-07T10:57:30.804332Z","submitted_at":"2025-06-04T04:32:54Z","title":"MiMo-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.03569","snapshot_observed_at":"2026-08-05T23:03:10.109377Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.109377Z"},"links":{"cited_paper":"/paper/2506.03569","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:6b096e87c1f6ff86f14a3bb6b56b35080c4983c0608bc247f625a08847344818","observation_id":"71aee071-de03-4200-b863-2ceaaa125bea","resolution":{"observed_at":"2026-08-05T23:03:10.109377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19786","last_updated":"2025-03-25T15:52:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-25T15:52:34Z","title":"Gemma 3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19786","snapshot_observed_at":"2026-08-05T23:03:10.112701Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.112701Z"},"links":{"cited_paper":"/paper/2503.19786","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:5198d301263658ffe0cea6c88e91b1eb78c1177649e729b12aa1abd0acf43e27","observation_id":"2ea5c41e-9f20-4505-9e8e-01755079951f","resolution":{"observed_at":"2026-08-05T23:03:10.112701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07491","last_updated":"2025-06-23T13:45:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-10T06:48:26Z","title":"Kimi-VL Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07491","snapshot_observed_at":"2026-08-05T23:03:10.115570Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.115570Z"},"links":{"cited_paper":"/paper/2504.07491","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:7212e02b24176dd64f2df4246dd3a6e9dbf0124da60630cf52d81a8fa44166ea","observation_id":"5e485052-805b-4905-8a77-1f30ab78ac14","resolution":{"observed_at":"2026-08-05T23:03:10.115570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01949","last_updated":"2025-07-02T17:57:28Z","snapshot_observed_at":"2026-08-06T20:37:19.866170Z","submitted_at":"2025-07-02T17:57:28Z","title":"Kwai Keye-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.01949","snapshot_observed_at":"2026-08-05T23:03:10.118939Z","title":"K.; Yang, B.; Wen, B.; Liu, C.; Chu, C.; Song, C.; Rao, C.; Yi, C.; Li, D.; Zang, D.; et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.118939Z"},"links":{"cited_paper":"/paper/2507.01949","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:5bbf1dbd2b0eaaf0b95891b3562846e88a98378c2dafdfa511da0e705ffd97c1","observation_id":"1e0d2528-1958-4e3f-b77f-1efc4f76eaf6","resolution":{"observed_at":"2026-08-05T23:03:10.118939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08837","last_updated":"2025-05-08T06:35:06Z","snapshot_observed_at":"2026-07-06T21:08:05.749656Z","submitted_at":"2025-04-10T17:41:56Z","title":"VL-Rethinker: Incentivizing Self-Reflection of Vision-Language Models with Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.08837","snapshot_observed_at":"2026-08-05T23:03:10.122029Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.122029Z"},"links":{"cited_paper":"/paper/2504.08837","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:b6381e1f4896af6eed2d796920a4fe2846ffb31665cce77d4d2fce3b6ef6bd7f","observation_id":"618dd905-d1d5-49ec-99d7-26c90ea45130","resolution":{"observed_at":"2026-08-05T23:03:10.122029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.917486Z","title":null,"venue":null,"work_id":"8154a0d2-86f2-4f92-b898-088257f08681","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.125020Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:385c136dcb81ff909e18db00247a838fcf8672fca483461993e3803fbb7a29d7","observation_id":"260dc3eb-3791-42c0-8cfe-437c4713d715","resolution":{"observed_at":"2026-08-05T23:03:10.920534Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.908329Z","title":null,"venue":null,"work_id":"0e820ccd-7147-4cde-9702-63e50e45200a","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.128111Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:a5cbcb40cc37ed80aa4cf291858dd531b110b0ebaf249c70b34ccbdf0179b055","observation_id":"3871fa7b-7502-439a-94b1-7fcaea51847c","resolution":{"observed_at":"2026-08-05T23:03:10.911319Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.899569Z","title":null,"venue":null,"work_id":"c28e70f2-9c9a-486d-a2b9-d3f0127c09ad","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.131052Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:650d26ff534d3eb2a385bd85aafb52f459ed1a94f552b8c44e942caba8918baf","observation_id":"2a694aaf-8b1a-4551-b2ec-17d8ecf284fb","resolution":{"observed_at":"2026-08-05T23:03:10.902563Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07934","last_updated":"2025-05-30T15:53:16Z","snapshot_observed_at":"2026-08-07T16:06:46.096701Z","submitted_at":"2025-04-10T17:49:05Z","title":"SoTA with Less: MCTS-Guided Sample Selection for Data-Efficient Visual Reasoning Self-Improvement","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07934","snapshot_observed_at":"2026-08-05T23:03:10.133767Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.133767Z"},"links":{"cited_paper":"/paper/2504.07934","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:268b9ffca8d8e6526bdb1c5b130f17132ee0761b8ada881e5c19308f6bf9012f","observation_id":"f055fe9b-4553-40eb-bc57-d58d410b375a","resolution":{"observed_at":"2026-08-05T23:03:10.133767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.137008Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.137008Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:b44ee3a97ef386650da313075b4ace2d135bdc0e9047d1acec13afc6ae858b3e","observation_id":"41cbcac5-d37f-48cf-91ae-e0388f1b2d00","resolution":{"observed_at":"2026-08-05T23:03:10.137008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18589","last_updated":"2025-05-13T06:32:02Z","snapshot_observed_at":"2026-08-07T15:59:38.409028Z","submitted_at":"2025-04-24T06:16:38Z","title":"Benchmarking Multimodal Mathematical Reasoning with Explicit Visual Dependency","version":4},"cited_work":{"arxiv_id":"2504.18589","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.18589","snapshot_observed_at":"2026-08-05T23:03:10.377640Z","title":"Benchmarking Multimodal Mathematical Reasoning with Explicit Visual Dependency","venue":"cs.CV","work_id":"e4e60d57-ccef-4829-915f-51854a152825","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.139666Z"},"links":{"cited_paper":"/paper/2504.18589","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:9b34dc6a679c4fae8c3d5ccd5b622866c1985350cc3370c3fd2dd51b8292b34d","observation_id":"1bd96295-23af-4c3e-a5db-9dc5ee03bd66","resolution":{"observed_at":"2026-08-05T23:03:10.382590Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.142900Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.142900Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:1590649cf951896d08cd4092d87d2a9e278b4b11f346cc47695bb4d6cf130de6","observation_id":"4fb2e29a-2ccb-4639-9c21-9808a76826c5","resolution":{"observed_at":"2026-08-05T23:03:10.142900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.890649Z","title":null,"venue":null,"work_id":"ff25a9b4-a668-4a51-bddc-be314edc3af5","year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.146186Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:4b9ce58a5ca57d69e54182b0bd4bad6a18700fca3fce3e11e6d54348709aae73","observation_id":"9ce07fde-0d99-462f-9dcc-29e2e4c88fe9","resolution":{"observed_at":"2026-08-05T23:03:10.893404Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04973","last_updated":"2024-07-06T06:48:16Z","snapshot_observed_at":"2026-07-06T18:42:18.289966Z","submitted_at":"2024-07-06T06:48:16Z","title":"LogicVista: Multimodal LLM Logical Reasoning Benchmark in Visual Contexts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04973","snapshot_observed_at":"2026-08-05T23:03:10.149147Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.149147Z"},"links":{"cited_paper":"/paper/2407.04973","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:55a634984543dbb4198397da2b72bc65f38a661053ab5f609c639fcccdb646a0","observation_id":"a7d2f13b-9698-4f37-8ec8-bb2072bd816f","resolution":{"observed_at":"2026-08-05T23:03:10.149147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11819","last_updated":"2024-02-02T02:35:13Z","snapshot_observed_at":"2026-07-06T17:18:40.469815Z","submitted_at":"2024-01-22T10:30:11Z","title":"SuperCLUE-Math6: Graded Multi-Step Math Reasoning Benchmark for LLMs in Chinese","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11819","snapshot_observed_at":"2026-08-05T23:03:10.152722Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.152722Z"},"links":{"cited_paper":"/paper/2401.11819","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:2c54253dac825f706a0806e6ca137b10c8fa9eb8eb84c1f9cc74a3fc0794162b","observation_id":"d42b273b-e04a-4d0d-b3fa-ccdea184d038","resolution":{"observed_at":"2026-08-05T23:03:10.152722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-05T23:03:10.155851Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.155851Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:820f5608658c7f6d81fdf3c69d405bd5907647ba2d0907186fcba3902db33423","observation_id":"ac6f4c5d-f2f3-4688-ac28-1377088bed02","resolution":{"observed_at":"2026-08-05T23:03:10.155851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07905","last_updated":"2025-06-09T16:20:54Z","snapshot_observed_at":"2026-08-07T05:20:11.340991Z","submitted_at":"2025-06-09T16:20:54Z","title":"WeThink: Toward General-purpose Vision-Language Reasoning via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07905","snapshot_observed_at":"2026-08-05T23:03:10.158919Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.158919Z"},"links":{"cited_paper":"/paper/2506.07905","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:0a927fc68c50c863b747d3706ad9c269015ed31ef7a52493826cc51ac34101b9","observation_id":"81f9e82d-4567-4976-9de0-b0ba9a4a3bb7","resolution":{"observed_at":"2026-08-05T23:03:10.158919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-07-06T20:00:47.730252Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-05T23:03:10.162023Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.162023Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:5e6985b6d9213c184c5079100a55a9b1ba1d1a439d5f12047ce3f4217bf0f8d7","observation_id":"0b2729ac-0c6d-42a0-a2a7-401f20bcb0e0","resolution":{"observed_at":"2026-08-05T23:03:10.162023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04550","last_updated":"2025-03-06T15:36:06Z","snapshot_observed_at":"2026-08-07T17:24:22.454511Z","submitted_at":"2025-03-06T15:36:06Z","title":"Benchmarking Reasoning Robustness in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.04550","snapshot_observed_at":"2026-08-05T23:03:10.164969Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.164969Z"},"links":{"cited_paper":"/paper/2503.04550","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:61314afa611110ec779b8ed4feb4c9827479b687478d74951072d3813c9e8c11","observation_id":"ab2b6813-c8b3-4b9b-b62b-dc614589eade","resolution":{"observed_at":"2026-08-05T23:03:10.164969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.881857Z","title":null,"venue":null,"work_id":"bc54122b-734c-4f43-9366-cfa9f9f6ecb8","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.167996Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:4214dc090d1a25f0fe40db7bfdbb003c63512142bcdb9eab05eac861b138fd57","observation_id":"91144bb5-3f36-47df-adb1-01aa8ccf59ce","resolution":{"observed_at":"2026-08-05T23:03:10.885047Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T23:03:10.872277Z","title":null,"venue":null,"work_id":"e432f833-4fc5-4a48-aa1c-587496ced770","year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.170967Z"},"links":{"citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:2b6d0c4fd4747e83ab4f0bce5bd5fb647aa9765a9d1c2949f23bfa4ea7307a34","observation_id":"3547a3a2-b23f-4526-acf2-6c9d0fb60d39","resolution":{"observed_at":"2026-08-05T23:03:10.875416Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08100","last_updated":"2024-06-12T11:27:03Z","snapshot_observed_at":"2026-07-06T18:29:30.075334Z","submitted_at":"2024-06-12T11:27:03Z","title":"Multimodal Table Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08100","snapshot_observed_at":"2026-08-05T23:03:10.173220Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.173220Z"},"links":{"cited_paper":"/paper/2406.08100","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:1beb0f2a6d583b51577938f79b97db3dad9bccde3156be4711dcdc292d9744c3","observation_id":"1cf70cc8-b296-4921-858d-26877ae6012a","resolution":{"observed_at":"2026-08-05T23:03:10.173220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-05T23:03:10.175885Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.175885Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:0c53d3bc4307f44c3ee81ff4c8a1ce400d45fefb9e7541c8755e69994b0a6d09","observation_id":"c7cda48d-4955-4a87-b877-5afcf3d4c5f3","resolution":{"observed_at":"2026-08-05T23:03:10.175885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00836","last_updated":"2025-02-24T06:55:22Z","snapshot_observed_at":"2026-07-06T19:43:46.557944Z","submitted_at":"2024-10-29T17:29:19Z","title":"DynaMath: A Dynamic Visual Benchmark for Evaluating Mathematical Reasoning Robustness of Vision Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00836","snapshot_observed_at":"2026-08-05T23:03:10.178730Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.178730Z"},"links":{"cited_paper":"/paper/2411.00836","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:0cce66d5a746c36c289dc7b197320bec4223f8cf9cce7dc8f87cc0fc38bedef7","observation_id":"803706d1-06d3-4b48-8f87-64fdea209998","resolution":{"observed_at":"2026-08-05T23:03:10.178730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T16:03:13.511884Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models"},"reference_resolution":{"displayed":70,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":65,"verified_exact":2,"verified_fuzzy":3},"total_outbound_references":70},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 70 of 70 outbound references and 0 inbound Pith citation observations for arXiv:2508.06009."}