{"as_of":"2026-08-06T01:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8d2ee1c6998b95ed8b77b51a90451b466d3dda1d9553a232ef33623b44543740","coverage":[{"denominator":78,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":78,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-19T05:55:09.188048Z","state":"measured"},{"denominator":80,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":80,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T23:12:32.101737Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-06-28T23:12:46.506585Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"cited_work":{"arxiv_id":"2507.01955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.01955","snapshot_observed_at":"2026-06-28T23:12:46.506585Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","venue":"cs.CV","work_id":"d5132995-7b29-49f6-bdf7-c4cc201187b0","year":2025},"citing_paper":{"arxiv_id":"2604.21346","last_updated":"2026-04-23T07:03:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-23T07:03:48Z","title":"Symbolic Grounding Reveals Representational Bottlenecks in Abstract Visual Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-09T22:24:44.884495Z"},"links":{"cited_paper":"/paper/2507.01955","citing_paper":"/paper/2604.21346"},"observation_digest":"sha256:01deb9d4ebe74f5803536a9c92ccf9e1e530337553a05a6a9194c8264710985e","observation_id":"f10f806c-317f-4939-8b37-230e2c825177","resolution":{"observed_at":"2026-05-09T22:39:15.088967Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"cited_work":{"arxiv_id":"2507.01955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.01955","snapshot_observed_at":"2026-06-28T23:12:46.506585Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","venue":"cs.CV","work_id":"d5132995-7b29-49f6-bdf7-c4cc201187b0","year":2025},"citing_paper":{"arxiv_id":"2605.31212","last_updated":"2026-05-29T12:18:08Z","snapshot_observed_at":"2026-07-06T23:40:22.583302Z","submitted_at":"2026-05-29T12:18:08Z","title":"Benchmarking and Enhancing Text-to-Image Models for Generating Visual Representations in Early Arithmetic Education","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T23:12:32.101737Z"},"links":{"cited_paper":"/paper/2507.01955","citing_paper":"/paper/2605.31212"},"observation_digest":"sha256:ff917a95a5f87939d5734ba809a53f5e18f4a76c904240cbe402d62158841283","observation_id":"15d0897b-ccfe-4f79-a439-0d17abd1a35f","resolution":{"observed_at":"2026-06-28T23:12:46.508029Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.01955/citation-record","integrity":"/paper/2507.01955/integrity","json":"/paper/2507.01955/citation-record.json","paper":"/paper/2507.01955"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Slic superpixels compared to state-of-the-art superpixel methods","venue":null,"work_id":"be1e36c1-0e7e-43ef-9e94-9854d4f2a6f3","year":2012},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:0a685688f54f913a818fd40a8d45aec74dc56b6d8d10b034de7e56262919dde5","observation_id":"af1fc42c-2f6e-4ea4-93bc-781bf5de67c6","resolution":{"observed_at":"2026-05-19T06:13:00.633297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:f91aa36f97f33a6ffdc852aaf4cdd1b2276191f716419f7809c8b6b0ff2abc6c","observation_id":"4c552708-69fc-4795-a042-9e9a2c03f8c4","resolution":{"observed_at":"2026-05-19T05:57:08.152595Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The llama 3 herd of models","venue":null,"work_id":"7c307f3b-b5aa-46a4-b821-4af8ffca32ec","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:554a3b72447cf625f696c7ea50f97f4a03160ea660d23654134c24be303faa39","observation_id":"9d644d0d-8561-4e14-a800-6aca7549da9c","resolution":{"observed_at":"2026-05-19T06:13:00.636410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04810","last_updated":"2024-08-09T01:41:05Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T01:41:05Z","title":"UniBench: Visual Reasoning Requires Rethinking Vision-Language Beyond Scaling","version":1},"cited_work":{"arxiv_id":"2408.04810","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.04810","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unibench: Visual reasoning requires rethinking vision- language beyond scaling","venue":null,"work_id":"f885dbdc-2c09-404f-a8af-eedf200e2479","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2408.04810","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:61d9482565743ccb94c5a3004a9ad6f3a6d55a6415a6883754857b2c65c06716","observation_id":"cc4905d7-0b8d-4292-9cb9-a399ab984a99","resolution":{"observed_at":"2026-05-19T05:57:08.159043Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"31e3af5c-9fec-43d9-b533-5bb70172dd15","year":null},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:b15f84cc6ea32a989bcc072cd90b38b7b441ea5fd29afd9ac04d64e4563aeca5","observation_id":"76698047-1279-4ede-9909-1ff84911fde7","resolution":{"observed_at":"2026-05-19T06:13:00.630048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing claude 3.5 sonnet","venue":null,"work_id":"062bfff7-55e6-4449-93f9-305563e81887","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:ded059606f8dda4c832b28dfcc75ebc43fa2ad1802a00c2e5a4f3e630d24e6ce","observation_id":"796457a1-fa09-4710-a41e-c413d6899df4","resolution":{"observed_at":"2026-05-19T06:13:00.627137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09406","last_updated":"2024-06-14T14:43:26Z","snapshot_observed_at":"2026-07-06T18:30:32.982860Z","submitted_at":"2024-06-13T17:59:42Z","title":"4M-21: An Any-to-Any Vision Model for Tens of Tasks and Modalities","version":2},"cited_work":{"arxiv_id":"2406.09406","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.09406","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"4m-21: An any-to-any vision model for tens of tasks and modalities","venue":null,"work_id":"fd743528-d309-4580-b78f-3266ded9b266","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2406.09406","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:987cc8ea2311f36207c8e51c2ab040ce799a168df2a97a8ef2f0e8b06cd549b5","observation_id":"f4f32294-4695-4628-aaf1-deb3244d8dec","resolution":{"observed_at":"2026-05-19T05:57:08.142513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:8e74afaa10dc7aaa6f6ac2a28a7e9ce13e91e98fa65885fa8906551b6badb28e","observation_id":"849e7cf6-7fe5-4925-a38a-ed56f0d71f87","resolution":{"observed_at":"2026-05-19T05:57:08.137262Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07726","last_updated":"2024-10-10T17:28:23Z","snapshot_observed_at":"2026-08-05T10:06:10.880743Z","submitted_at":"2024-07-10T14:57:46Z","title":"PaliGemma: A versatile 3B VLM for transfer","version":2},"cited_work":{"arxiv_id":"2407.07726","doi":"10.48550/arxiv.2407.07726","metadata_source":"pith","pith_arxiv_id":"2407.07726","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PaliGemma: A versatile 3B VLM for transfer","venue":"cs.CV","work_id":"df6f48b3-5792-47c7-9614-cb856ea31ad9","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2407.07726","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:ca3b971c460182f50a8a42487a3417250a4b1cf536e6e994cd6d98d7cda81d23","observation_id":"dd9afcaa-8d04-42c8-8195-84a29332c955","resolution":{"observed_at":"2026-05-19T05:57:08.147550Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-22T13:22:31.538292+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T13:22:31.538292+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omni3d: A large benchmark and model for 3d object detection in the wild","venue":null,"work_id":"090094f9-5be8-4c9a-a701-76b59e0b4fc1","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:94e850470a1c915344f1cffa1f3a7ca55f599e5aea3c95773029d7026f53a2d7","observation_id":"2857d08f-9d20-46ea-ba78-b043d324ede8","resolution":{"observed_at":"2026-05-19T06:13:00.734208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"End-to- end object detection with transformers","venue":null,"work_id":"dd9ce779-8619-40f5-bdaf-3d21a4b0011e","year":2020},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:3b1614e74975b7df67656cc89123d65ac9fc9d88bded503a94903d321ce943f8","observation_id":"925f2404-1be5-44df-a0cb-75b8b1d9e772","resolution":{"observed_at":"2026-05-19T06:13:00.737167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:df973be3700de51d984afe1618ee80995853a1c2b45278313658fd304154359b","observation_id":"d8337363-346b-47ab-b363-adc6f363a953","resolution":{"observed_at":"2026-05-19T05:57:08.070966Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05979","last_updated":"2025-04-10T18:02:00Z","snapshot_observed_at":"2026-07-06T21:06:08.430339Z","submitted_at":"2025-04-08T12:34:36Z","title":"An Empirical Study of GPT-4o Image Generation Capabilities","version":2},"cited_work":{"arxiv_id":"2504.05979","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.05979","snapshot_observed_at":"2026-07-04T13:49:51.431619Z","title":"An empirical study of gpt-4o image generation capabilities","venue":null,"work_id":"8ae5e60a-7e08-491f-bea0-f4432ea5f6e6","year":2025},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2504.05979","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:024350f7ff88f22257a2a19ddcb4bde191647de3a0e17f9578f47fa6718b7b1f","observation_id":"eb24732f-0bf0-418c-b2d8-fe4667ea7b09","resolution":{"observed_at":"2026-05-19T05:57:08.065260Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-icl: Zero-shot in-context learning with self- generated demonstrations","venue":null,"work_id":"7755de31-6b64-4dba-b393-ba8fcdd3aef8","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:95ddf1fe24e226174a7e7b4d96579748c5e0b2a35b4903010cec1df2a60aa160","observation_id":"2ee451af-a98f-4a60-be4a-ed43905084ed","resolution":{"observed_at":"2026-05-19T06:13:00.711639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reproducible scal- ing laws for contrastive language-image learning","venue":null,"work_id":"1dac5015-e675-4850-9eb0-c80360a35759","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:7aa5fd9f3f46c99cb83211e53ae41588aeed04f2c0f92b88d90bddb1a7a64dc6","observation_id":"a5a53400-4bc9-47b3-a2b8-01c4f7ac540a","resolution":{"observed_at":"2026-05-19T06:13:00.728328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:a44637557c980bc5ffb0f72173f26c32ee8c14db95404325c9f452b95135e47d","observation_id":"5c39fabb-892a-4e64-8eb7-441385c8b786","resolution":{"observed_at":"2026-05-19T05:57:08.076312Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.09670","last_updated":"2021-10-31T20:03:39Z","snapshot_observed_at":"2026-07-06T10:05:50.314212Z","submitted_at":"2020-10-19T17:06:18Z","title":"RobustBench: a standardized adversarial robustness benchmark","version":3},"cited_work":{"arxiv_id":"2010.09670","doi":"10.48550/arxiv.2010.09670","metadata_source":"arxiv_reference","pith_arxiv_id":"2010.09670","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Robustbench: a stan- dardized adversarial robustness benchmark","venue":"arXiv (Cornell University)","work_id":"8ae4b2b2-a2da-4900-9021-ad64ae1b860f","year":2010},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2010.09670","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:9f2227bdcc3c876c0b7d1765c4db244e628c54c38703e74227ae2b37741ba54c","observation_id":"0d4ad231-3973-4dce-ad72-6f8e6e600c80","resolution":{"observed_at":"2026-05-19T05:57:08.095871Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InstructBLIP: Towards general-purpose vision-language models with instruction tuning","venue":null,"work_id":"154002f8-0780-4c5a-b5dc-43e3396fd0ae","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:d048568e159778707ea0d510e8897f2a76d95bb7eda71e14390247749ba16698","observation_id":"95dcfcdb-207a-4700-a145-00ca4333cc3c","resolution":{"observed_at":"2026-05-19T06:13:00.705908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omnidata: A scalable pipeline for making multi-task mid-level vision datasets from 3d scans","venue":null,"work_id":"6bcee5d5-44f8-4e70-98cc-20a3b8cb8d68","year":2021},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:6436a55c2b562aeda4896d49145a258af63e41fda6298e0c97cc1aadd7b3ec60","observation_id":"730fb3d8-ed13-4f0e-8647-67f0203e295b","resolution":{"observed_at":"2026-05-19T06:13:00.699880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Find your inspiration","venue":null,"work_id":"b1f23790-f6ce-4fee-ab1a-1647d369412e","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:e401013799be75d4a48c35e886ed39f64f3079b30e2f638f40c82ac505fcfed1","observation_id":"c0671321-6117-43a8-ab20-b43f891fbbbb","resolution":{"observed_at":"2026-05-19T06:13:00.724740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12390","last_updated":"2024-07-03T08:44:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-18T17:59:54Z","title":"BLINK: Multimodal Large Language Models Can See but Not Perceive","version":4},"cited_work":{"arxiv_id":"2404.12390","doi":"10.1080/15732479.2026.2630123","metadata_source":"pith","pith_arxiv_id":"2404.12390","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BLINK: Multimodal Large Language Models Can See but Not Perceive","venue":"cs.CV","work_id":"10e71374-f61b-477c-9433-12b7990c388e","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2404.12390","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:8a66ff98b623ba787f79aa8552017a7cb57fafa3680dc652dfcbece14961146e","observation_id":"84230edf-4209-4308-9141-436bcd19977b","resolution":{"observed_at":"2026-05-19T05:57:08.011869Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Explore vision capabilities with the gemini api","venue":null,"work_id":"8c54984d-6e93-4c43-968b-bdae2a7082b4","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:eedf6dc377692da9e7c0170afda6849c5dc4289d3fa1ecf24f7a69c156d13745","observation_id":"c79d67e7-4ba0-4d8c-8528-915765975396","resolution":{"observed_at":"2026-05-19T06:13:00.731642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini 2.0 flash","venue":null,"work_id":"c674db0f-908b-4632-86a7-9cdcc1ab2e20","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:b3b88d9c5f444dc09a04b8b01a9179bc76977eecc262e35f79724865fb13ebd2","observation_id":"ade1cdb5-0570-4e03-a98d-7f347cf903ee","resolution":{"observed_at":"2026-05-19T06:13:00.740029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1903.12261","last_updated":"2019-03-28T20:56:37Z","snapshot_observed_at":"2026-08-02T09:01:28.869881Z","submitted_at":"2019-03-28T20:56:37Z","title":"Benchmarking Neural Network Robustness to Common Corruptions and Perturbations","version":1},"cited_work":{"arxiv_id":"1903.12261","doi":"10.48550/arxiv.1903.12261","metadata_source":"pith","pith_arxiv_id":"1903.12261","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Benchmarking Neural Network Robustness to Common Corruptions and Perturbations","venue":"cs.LG","work_id":"e28eaed9-7f6b-46ed-ba19-fba407571ba7","year":2019},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/1903.12261","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:dd53bc82872888e2f0578259b0a714aaa1e8256813e655b098acd650253554a5","observation_id":"6bbb666f-1d9f-4f5d-b1eb-55311d29b474","resolution":{"observed_at":"2026-05-19T05:57:08.007151Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":"2009.03300","doi":"10.48550/arxiv.2009.03300","metadata_source":"pith","pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Massive Multitask Language Understanding","venue":"cs.CY","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","year":2020},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:e86df467b15bf274158786d9f3bd086a365b15ca5b6a4de82e43d36c1dadbd46","observation_id":"df572c0c-a2e5-4fd4-86d2-271579942874","resolution":{"observed_at":"2026-05-19T05:57:08.111288Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The many faces of robust- ness: A critical analysis of out-of-distribution generalization","venue":null,"work_id":"df78c6ff-0ce1-479e-9b26-7db91f76969b","year":2021},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:e8a543b6920c168d4b510a2cf36051da8fe8474dbf1a079c61aeb23a74f93d68","observation_id":"3236a1d4-eee7-4b55-8888-743ea23c65d6","resolution":{"observed_at":"2026-05-19T06:13:00.742942Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09403","last_updated":"2024-11-11T00:54:32Z","snapshot_observed_at":"2026-08-05T09:57:39.370124Z","submitted_at":"2024-06-13T17:59:31Z","title":"Visual Sketchpad: Sketching as a Visual Chain of Thought for Multimodal Language Models","version":3},"cited_work":{"arxiv_id":"2406.09403","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.09403","snapshot_observed_at":"2026-07-04T19:30:06.763408Z","title":"Visual sketchpad: Sketching as a visual chain of thought for multimodal language models","venue":null,"work_id":"6df227a2-7a70-47c8-ae06-b2d6eee27a57","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2406.09403","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:f526921d27f26d022b458438b43cfe78ccd15825f44550dd314711b0c6021d1d","observation_id":"7ead8605-e0da-4ea8-a98c-dd2ddaf30320","resolution":{"observed_at":"2026-05-19T05:57:08.017947Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llama multiple images","venue":null,"work_id":"432ac266-5822-47d2-8047-7c1503d32be8","year":null},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:06cb48a095292c130c823d41a73903744160002f9e85695398202dd0f2785a7d","observation_id":"02e8f7a4-7cd6-4b78-8421-92cceb6dec0c","resolution":{"observed_at":"2026-05-19T06:13:00.745902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10160","last_updated":"2023-10-18T13:17:13Z","snapshot_observed_at":"2026-07-06T15:28:40.452146Z","submitted_at":"2023-05-17T12:23:38Z","title":"Stop Uploading Test Data in Plain Text: Practical Strategies for Mitigating Data Contamination by Evaluation Benchmarks","version":2},"cited_work":{"arxiv_id":"2305.10160","doi":"10.48550/arxiv.2305.10160","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.10160","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Stop uploading test data in plain text: Practical strategies for mitigating data contamination by evaluation benchmarks","venue":"arXiv (Cornell University)","work_id":"fc2502a0-2c44-4a07-9561-fb571d2d5d4f","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2305.10160","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:7ff3b2a9d44c9b6243d681e77a3c32e4a8c4f1ac69d27465de491cf771d4175e","observation_id":"8948bcd2-cb98-4b5b-a743-8de1248f04a9","resolution":{"observed_at":"2026-05-19T05:57:07.979885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Oneformer: One transformer to rule universal image segmentation","venue":null,"work_id":"32e3e323-f4f0-4990-8255-cdc59382122c","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:e78fb58aefd60f0a11edb341e05b9dea9716be250f7837623728c6126172dfff","observation_id":"0050aef9-85f8-41d0-b6fb-84e79ea0f2d6","resolution":{"observed_at":"2026-05-19T06:13:00.758433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09798","last_updated":"2024-10-04T21:51:47Z","snapshot_observed_at":"2026-07-06T18:15:03.339204Z","submitted_at":"2024-05-16T04:02:43Z","title":"Many-Shot In-Context Learning in Multimodal Foundation Models","version":2},"cited_work":{"arxiv_id":"2405.09798","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.09798","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chen, and An- drew Y","venue":null,"work_id":"8ee3d531-9223-41d0-9911-fabfee119f94","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2405.09798","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:ce4c603032a380182a770b9bea607e7c2bdf3b915bb3550689fbb4abfd4043c3","observation_id":"973035ca-77dd-43cd-8391-1e740949f05d","resolution":{"observed_at":"2026-05-19T05:57:08.083275Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chen, and Andrew Y","venue":null,"work_id":"91bf1087-d1a5-4df9-bb3a-ba2b2924b6d0","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:4e26d23e4941c5674989aa3ada8b17b42e76150909c6da3d6027d6564ef50406","observation_id":"b8a5010f-9722-4133-bb60-b2db95d3b2ec","resolution":{"observed_at":"2026-05-19T06:13:00.702966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"3d common corruptions and data augmentation","venue":null,"work_id":"e1c0fc19-0a86-4c2a-9d82-0292a3f639f7","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:02e8a1efd4d7abe0a201b1d530b1a23c7d3c17c942945c5c85ae906d9477236a","observation_id":"4108af43-0196-432a-8b3b-576dadfce6b7","resolution":{"observed_at":"2026-05-19T06:13:00.764254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"3d com- mon corruptions for object recognition","venue":null,"work_id":"c3f35490-7c4e-453d-80b9-0cd7ed624fcb","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:699c78c79898f8901d1c371de32f00ee61a197738befa6224b070b3afc22ea58","observation_id":"d3e3db48-7e1c-46c8-b4c3-c1a86f15a2ed","resolution":{"observed_at":"2026-05-19T06:13:00.748514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02406","last_updated":"2023-04-11T19:39:17Z","snapshot_observed_at":"2026-07-06T14:00:01.450718Z","submitted_at":"2022-10-05T17:28:20Z","title":"Decomposed Prompting: A Modular Approach for Solving Complex Tasks","version":2},"cited_work":{"arxiv_id":"2210.02406","doi":"10.48550/arxiv.2210.02406","metadata_source":"pith","pith_arxiv_id":"2210.02406","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Decomposed Prompting: A Modular Approach for Solving Complex Tasks","venue":"cs.CL","work_id":"33743159-d481-436a-8723-268747b8b495","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2210.02406","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:a4790cf6010b0d10c471ab0b57975d04ddd4265ee5c692d81fb562a1c1e2327e","observation_id":"036c8900-e6b2-4266-9847-e8efc8e45b8b","resolution":{"observed_at":"2026-05-19T06:14:05.332461Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02643","last_updated":"2023-04-05T17:59:46Z","snapshot_observed_at":"2026-07-06T15:12:37.953383Z","submitted_at":"2023-04-05T17:59:46Z","title":"Segment Anything","version":1},"cited_work":{"arxiv_id":"2304.02643","doi":"10.1109/iccv51070.2023.00371","metadata_source":"pith","pith_arxiv_id":"2304.02643","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Segment Anything","venue":"cs.CV","work_id":"2bbf46ca-720a-45a1-8e9c-10c33fbeada0","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2304.02643","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:ae241f6e4fe644e1dd9366e40866776d6d4a3df6f1692d21e2c66d4fd51b8844","observation_id":"2c62bba2-d900-44ba-aeee-2bb0cc87b4bb","resolution":{"observed_at":"2026-05-19T05:57:08.132363Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:58.489975+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:58.489975+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":"2301.12597","doi":"10.48550/arxiv.2301.12597","metadata_source":"pith","pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","venue":"cs.CV","work_id":"63d03f4d-15f4-4583-8286-913c19f02294","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:62b8b5c1b95af7ab31f6ffbe036515d53c93a3b875e915689f2fbc5fd7f188dc","observation_id":"87106f7e-ac80-4372-9227-e5d629d9cbbf","resolution":{"observed_at":"2026-05-19T05:57:07.985092Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10355","last_updated":"2023-10-26T02:52:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T16:34:01Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2305.10355","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.10355","snapshot_observed_at":"2026-07-10T11:37:03.198858Z","title":"Evaluating Object Hallucination in Large Vision-Language Models","venue":"cs.CV","work_id":"66d8ac3e-c134-4995-b528-550afa17586f","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2305.10355","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:cc606933c6224ee4c997bd3becbe4d8a76da642a80d21e21dcbabe852c18ea29","observation_id":"d8bfc312-e3fa-462e-8b82-cc85b5c18280","resolution":{"observed_at":"2026-05-19T05:57:08.046330Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Dollár, and C","venue":null,"work_id":"123c9edf-7a91-43cb-a300-748bbf56f69d","year":2014},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:b3679737ca889779a831a14a8e3d290fe2942d040830e32536b81eb8cb3d5b4e","observation_id":"31b85b5e-3e4a-4ca4-a6e6-8d011e5bd977","resolution":{"observed_at":"2026-05-19T06:13:00.714845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llava-next: Improved reason- ing, ocr, and world knowledge","venue":null,"work_id":"60f9556e-1945-4c36-902d-f38e23fb27e2","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:610049ca17199d7d4ca2e78c24430154b95677fce38241602b7c2a010a9d67db","observation_id":"b2633707-72e5-45a9-a7cd-99ca8e7ceba8","resolution":{"observed_at":"2026-05-19T06:13:00.691806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llama 90b","venue":null,"work_id":"e0f32e85-dd50-416c-b73e-fd7a2005c756","year":null},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:309f521d69daefcb4cd79c81c39facb7a463a3c85cd23a2ac5cc40f17296e6c2","observation_id":"b60ef790-8b20-4ad7-96ea-a956880e5b0b","resolution":{"observed_at":"2026-05-19T06:13:00.694330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"4M: Massively multimodal masked modeling","venue":null,"work_id":"174524c7-1434-4930-addf-7b601fe7c20d","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:abb3a5470696b90e208ae44e409489e6ec520af2d9227c17cc53518154997f0c","observation_id":"5f21c826-b6a9-46b9-adb4-1b22d7d3983c","resolution":{"observed_at":"2026-05-19T06:13:00.720188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing openai o1","venue":null,"work_id":"49fd2cf2-8cc5-4f93-ba67-167af3185ef8","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:e014602d1192741c8d118713eaab2a72619cf91f7c5c16d659343410f90a6247","observation_id":"caa1614e-ebf5-4ff0-9589-84e51b30f0ec","resolution":{"observed_at":"2026-05-19T06:13:00.689259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hello gpt-4o","venue":null,"work_id":"358b8d94-fdb9-4a17-a31f-08e2f47ac8cf","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:41a28fd808587ada40f3a0b41537a0eca9d9d282e5759915cb489d4e38664192","observation_id":"8d810ea7-01d5-48ed-bba2-64f380570a8e","resolution":{"observed_at":"2026-05-19T06:13:00.755078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing 4o image generation","venue":null,"work_id":"01e9ec01-9160-4d96-b426-530dd8f6d708","year":2025},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:1980049a432f57a369835848204534a54fe6fb746cc5627297694ea1859e8be5","observation_id":"bde9b641-06a3-4a78-9d7e-5f65a9716d32","resolution":{"observed_at":"2026-05-19T06:13:00.685707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing openai o3 and o4-mini","venue":null,"work_id":"0660a84e-462d-4941-b04d-db9978819231","year":2025},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:3f0161864d825448b5faeb8ea3bf16c93c1917cad18f8c38a2820c736d8c3ad8","observation_id":"e7521322-5fc0-41de-8eb0-ad9449d1db3e","resolution":{"observed_at":"2026-05-19T06:13:00.682787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06581","last_updated":"2025-03-27T16:16:09Z","snapshot_observed_at":"2026-08-03T13:03:43.962537Z","submitted_at":"2024-07-09T06:20:17Z","title":"Vision language models are blind: Failing to translate detailed visual features into words","version":6},"cited_work":{"arxiv_id":"2407.06581","doi":"10.48550/arxiv.2407.06581","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.06581","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vision language models are blind","venue":"arXiv (Cornell University)","work_id":"21fe7700-6786-4b12-a1dd-88d16b4403c2","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2407.06581","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:d3665fd5eafa26fea9d4a3735fb2149929731d4b9c2ce302d1286304245dafa1","observation_id":"8bd60bd1-6dd6-41cf-8065-10bf0bc2c12c","resolution":{"observed_at":"2026-05-19T05:57:07.991005Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Do imagenet classifiers generalize to im- agenet? In International conference on machine learning , pages 5389–5400","venue":null,"work_id":"4206a070-1d2f-4c74-8d32-aa154ffed74d","year":2019},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:9f7a758a1a8033cd54e82296b5521feacf05569339a099336f98d4409e99da0b","observation_id":"f276678e-f04e-4304-96e6-e1db2b28a8c7","resolution":{"observed_at":"2026-05-19T06:13:00.709224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:806a8cddce10799eba8f049b7586c83ab28e280407457454b6c1e670cb8a6b80","observation_id":"8be94be3-92eb-45b7-be9d-0d662b598277","resolution":{"observed_at":"2026-05-19T05:57:08.034625Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-04T22:55:15.345443Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":"2311.12022","doi":"10.48550/arxiv.2311.12022","metadata_source":"pith","pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","venue":"cs.AI","work_id":"9e2a976b-f5ad-4aee-af5c-243fe0fe75d2","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:6db2d8891ee81ac9c7331933a0dcde54f2f60bc3b6a8fa026904a319efb9f6f8","observation_id":"bfa972c0-e570-4f0d-86ef-a64c9340edf0","resolution":{"observed_at":"2026-05-19T05:57:08.028629Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-01T04:38:46.941438+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T04:38:46.941438+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hypersim: A photorealistic synthetic dataset for holistic indoor scene understanding","venue":null,"work_id":"6fe9ab17-9281-49ec-885b-7ce88fc6e331","year":2021},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:49bd402dee815538b6d07a0e92f25853d47dfd0bbee88d6be66f569812f800a0","observation_id":"3baaf630-7cc1-4ff2-8d78-9f99823937a8","resolution":{"observed_at":"2026-05-19T06:13:00.721474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bernstein, Alexander C","venue":null,"work_id":"18c60a47-c156-47e4-928c-2ce4338ce32d","year":2014},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:ec6af9a27ca66833934b708441f929d803d606678c1501282935a7a640b5a9ea","observation_id":"f59fb832-03c7-465f-9852-983109d65a38","resolution":{"observed_at":"2026-05-19T06:13:00.717995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Super- pixels: An evaluation of the state-of-the-art","venue":null,"work_id":"3fb100e0-c6cd-4f8a-b933-308ea8301274","year":2018},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:27631abb7cbd202122781921fbcfc25d76314efb4c289d9a93f6aa95456bb814","observation_id":"2a0a9a7d-cfb2-4517-9bf2-0bb3380aec2b","resolution":{"observed_at":"2026-05-19T06:13:00.761451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09818","last_updated":"2025-03-21T05:54:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-16T05:23:41Z","title":"Chameleon: Mixed-Modal Early-Fusion Foundation Models","version":2},"cited_work":{"arxiv_id":"2405.09818","doi":"10.48550/arxiv.2405.09818","metadata_source":"pith","pith_arxiv_id":"2405.09818","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chameleon: Mixed-Modal Early-Fusion Foundation Models","venue":"cs.CL","work_id":"2661b9a6-25cc-41a1-8100-612d2b801289","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2405.09818","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:ccbf07fbb6a555549c7d36692da7b52573912f4daf45af535fbb7e2d00ae5660","observation_id":"2b46d22b-83c9-4b4a-a96e-162b889799e0","resolution":{"observed_at":"2026-05-19T05:57:08.040162Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-10T17:49:22.013335+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T17:49:22.013335+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:8b1c92fa9a39f0f1d02f8a2f86204cb8dc9c523f5c9e1e8c164e3e44975ca509","observation_id":"a4b59d9c-d342-4c57-a817-5f8602939e42","resolution":{"observed_at":"2026-05-19T05:57:08.059358Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2406.16860","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.16860","snapshot_observed_at":"2026-07-03T23:49:02.295501Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","venue":"cs.CV","work_id":"f305b8a2-d78e-4bf3-b9fd-41777382e536","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2406.16860","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:52ef14087781630a4807d234e5146b01955375ea806b72b7c62dae5c12de1084","observation_id":"1f2549cc-fb7c-44bc-95a2-27ecc561455e","resolution":{"observed_at":"2026-05-19T05:57:07.996972Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:2a3d30237d2ba0d587acc0b83924d575cd79b829bf97554a6aa1fd0ba6ca4dce","observation_id":"ffedd751-f1f1-40a0-92bd-419d0354ae2e","resolution":{"observed_at":"2026-05-19T05:57:08.023724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The internet’s source for visuals","venue":null,"work_id":"0d5bf634-ef08-466f-bc68-17b0b34aa1a5","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:f0c463b4b156c4480ffc709cd62f3fc6239409c76e26ed310b4121c42b7bd0fc","observation_id":"5cfcc742-0105-4fbb-bced-2d8361878310","resolution":{"observed_at":"2026-05-19T06:13:00.677897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning robust global representations by penalizing local predictive power","venue":null,"work_id":"7c6c914f-b5f7-4c2c-be7a-fe09235818f5","year":2019},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:bfa8819023524ec5a489a2cea4fe750e9ef42243863e2af3e71bd4b6b4ddf0d0","observation_id":"056c4c45-ed18-4d98-a27d-5da1f431d869","resolution":{"observed_at":"2026-05-19T06:13:00.751780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14100","last_updated":"2022-12-15T19:21:35Z","snapshot_observed_at":"2026-08-04T09:31:34.784365Z","submitted_at":"2022-05-27T17:03:38Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","version":5},"cited_work":{"arxiv_id":"2205.14100","doi":"10.48550/arxiv.2205.14100","metadata_source":"pith","pith_arxiv_id":"2205.14100","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","venue":"cs.CV","work_id":"45b0563f-7563-4f00-8da5-8be70ff803fb","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2205.14100","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:e64a2b6091ab76fb061e01b072a9f77f5616399f553832017caafc15968dd5d6","observation_id":"5f5817ac-39fd-43d8-8f80-93b363ef8914","resolution":{"observed_at":"2026-05-19T05:57:08.002146Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-25T07:53:51.898843+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T07:53:51.898843+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":"2409.12191","doi":"10.48550/arxiv.2409.12191","metadata_source":"pith","pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","venue":"cs.CV","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:fc409369d00e66eb2e20e177a62d9e324d41499e5272939397fed8bcbf88b3ba","observation_id":"213f451e-94d2-463f-99cf-2529fedae323","resolution":{"observed_at":"2026-05-19T05:57:08.053030Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of- thought prompting elicits reasoning in large language models","venue":null,"work_id":"ec235e9b-2fbd-43fe-bc49-5c7d3978ef16","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:613ad00e0d17acbed9599aa76207ed14f48756717ae45418455961a7bb65c669","observation_id":"2ac40667-6aaa-441a-815a-faca4caf20a6","resolution":{"observed_at":"2026-05-19T06:13:00.675172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Morcos, Hongseok Namkoong, Ali Farhadi, Yair Carmon, Simon Kornblith, and Ludwig Schmidt","venue":null,"work_id":"6600b5c6-7422-413f-b9be-0fd3912b2f66","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:e3e87846aa07a4b9f779b581a16dd9799c3017a49ba645e949b81666a3e38640","observation_id":"ae4d69db-d967-4b4b-95a1-1a9f2762b48b","resolution":{"observed_at":"2026-05-19T06:13:00.686589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V*: Guided visual search as a core mechanism in multimodal llms","venue":null,"work_id":"08e8563c-dd25-4d62-a236-a07fd57d1b55","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:3d8b8fd24a70303e6280dd19639b62eb0f937c5ff4674bb03a6592d2d38d59c3","observation_id":"69876831-09f2-44a3-b370-55ab9781abd9","resolution":{"observed_at":"2026-05-19T06:13:00.672495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12488","last_updated":"2024-07-23T07:14:54Z","snapshot_observed_at":"2026-07-06T17:46:50.682231Z","submitted_at":"2024-03-19T06:54:33Z","title":"DetToolChain: A New Prompting Paradigm to Unleash Detection Ability of MLLM","version":3},"cited_work":{"arxiv_id":"2403.12488","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.12488","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dettoolchain: A new prompting paradigm to unleash detection ability of mllm","venue":null,"work_id":"35ca8613-a229-4b32-a63c-8176a0c39312","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2403.12488","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:622995651ce04002273a64e1f4177d72f2ea10e209f61789e7685a72cd1722e6","observation_id":"b5e57172-4c45-4398-9a9c-3339072f57b7","resolution":{"observed_at":"2026-05-19T05:57:08.122652Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11441","last_updated":"2023-11-06T07:39:49Z","snapshot_observed_at":"2026-08-04T03:29:49.409446Z","submitted_at":"2023-10-17T17:51:31Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","version":2},"cited_work":{"arxiv_id":"2310.11441","doi":"10.48550/arxiv.2310.11441","metadata_source":"pith","pith_arxiv_id":"2310.11441","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","venue":"cs.CV","work_id":"ddc8878e-ed0c-4a66-952a-254dce1c622a","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2310.11441","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:94e4740728578cca9c05eddb9e502b087c327dfeeeae98761aecee1124d891ee","observation_id":"a23c936f-d569-4cab-8394-1ceed439aaf2","resolution":{"observed_at":"2026-05-19T05:57:08.127733Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The dawn of lmms: Preliminary explorations with gpt-4v(ision)","venue":null,"work_id":"01321357-d05a-4277-b2ff-df86c3f72261","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:cc79331759d7237fb4ca818768460297a7a3f02192d60712f9771742ceb2e8ad","observation_id":"47e5d366-d3d8-48bd-94a6-4a69f495f551","resolution":{"observed_at":"2026-05-19T06:13:00.669221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tree of thoughts: Deliberate problem solving with large language models","venue":null,"work_id":"6df254c1-781f-4e42-9423-0c51ca8120f6","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:d6a2a7b4ed584e0fa947991d952b2eaa213559c76a4a78c4b9313f3c02e56452","observation_id":"0db5fbe9-c484-4b21-b977-62af7bc7cfb9","resolution":{"observed_at":"2026-05-19T06:13:00.680646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":"2306.13549","doi":"10.48550/arxiv.2306.13549","metadata_source":"pith","pith_arxiv_id":"2306.13549","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey on Multimodal Large Language Models","venue":"cs.CV","work_id":"29d41a4d-d51b-4020-8947-991367e75c2c","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2306.13549","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:6ab23450ff77ba8c1d2278af43848d12ffd13669e7272675998cc2511957231d","observation_id":"376d96c5-4802-4c08-8b5a-3b4ffcd3b0af","resolution":{"observed_at":"2026-05-19T05:57:08.106326Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mmmu: A massive multi-discipline multi- modal understanding and reasoning benchmark for expert agi","venue":null,"work_id":"4ba6351b-7763-4023-8e58-3fd6410e2ef2","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:18d7956c04f5aae7eb709710bbda0d33c7af593cb6af03138f6f7daae4358604","observation_id":"af8d3448-6315-4891-abf2-c2aab21c68cf","resolution":{"observed_at":"2026-05-19T06:13:00.665794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13601","last_updated":"2024-05-28T05:36:23Z","snapshot_observed_at":"2026-07-06T17:20:00.101661Z","submitted_at":"2024-01-24T17:10:45Z","title":"MM-LLMs: Recent Advances in MultiModal Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.13601","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.13601","snapshot_observed_at":"2026-07-04T15:09:55.103963Z","title":"Mm-llms: Recent ad- vances in multimodal large language models","venue":null,"work_id":"4c35990c-f0e6-4be5-bdd6-e77e3eaddf23","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2401.13601","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:ff4bd75801ba7684c2d8658d344efc0e13904b3d6667bf0da876df9d26d93024","observation_id":"822a7497-64f7-4414-9d5b-6cf82546187f","resolution":{"observed_at":"2026-05-19T05:57:08.117043Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scene parsing through ADE20K dataset","venue":null,"work_id":"e987f453-2116-439f-bf88-755f10c43e10","year":2017},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:332ee22cebdfd9a9b82ae57937e3160b98ab8f9cf67c3f012bc42064f1acf590","observation_id":"08aebd87-c1cb-4bf5-b47d-a4a3ffec27dc","resolution":{"observed_at":"2026-05-19T06:13:00.708902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.10625","last_updated":"2023-04-16T22:08:08Z","snapshot_observed_at":"2026-08-03T04:30:54.138138Z","submitted_at":"2022-05-21T15:34:53Z","title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","version":3},"cited_work":{"arxiv_id":"2205.10625","doi":"10.48550/arxiv.2205.10625","metadata_source":"pith","pith_arxiv_id":"2205.10625","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","venue":"cs.AI","work_id":"7e58c111-4666-4996-b5ad-1c8efd433083","year":2022},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2205.10625","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:23ce8e05fc077562a482eb5b0ad783a05662579db2ded527a1f32155d5f8181c","observation_id":"12eea387-8707-4645-84e3-3078f699f366","resolution":{"observed_at":"2026-05-19T05:57:08.101076Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Detrs with collab- orative hybrid assignments training","venue":null,"work_id":"19ecf78d-6f73-4056-a658-9315a5575bfe","year":2023},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:81e101d3370440bb78ceca08d3288f4ae40772e53a615ca4131556708d3b7b95","observation_id":"96d4ef64-4828-49f6-9958-e2f55a64254f","resolution":{"observed_at":"2026-05-19T06:13:00.674916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning ordinal relationships for mid-level vi- sion","venue":null,"work_id":"a793db25-ac80-459e-8159-a2466ff42e5d","year":2015},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:1c6ef09c73cf526726354d7decc38cce7e03fa775c3f04235f973867017a3cc9","observation_id":"a25b1049-d68a-4101-a9fa-985b3c5b141b","resolution":{"observed_at":"2026-05-19T06:13:00.683693Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"objectness","venue":null,"work_id":"037c9aeb-9586-40c8-b7f2-ca34c3ceb69e","year":null},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:72968cca95b03879fc6c4ae940cda7ab7b1aedb06a958906050c5ad15ce8a38e","observation_id":"b0615429-3242-496d-87d8-516a40145772","resolution":{"observed_at":"2026-05-19T06:13:00.653908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Produce a raw image (same dimensions as input) whose pixel colors encode the normals as above","venue":null,"work_id":"f8ab5cb0-714f-43fb-b5a3-546c8ea094be","year":null},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:91d8e947d6a4278112dcb1547a3cdcfc9fa256d3b9f36a8ce2a2ef73ccbbd18c","observation_id":"4da8a1cd-71eb-4304-a4b9-4ed8cf720607","resolution":{"observed_at":"2026-05-19T06:13:00.647136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Listing 3","venue":null,"work_id":"dcc6c0eb-e034-498e-83ac-8f88601a6b37","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:2e0d8d1a87c10017b31c9a3d21fdef5ba8d63329564e30304a4cfcc5b33c903d","observation_id":"3310b469-7a3f-417f-a91d-d20ba3c26157","resolution":{"observed_at":"2026-05-19T06:13:00.694038Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks"},"reference_resolution":{"displayed":78,"state_counts":{"malformed_identifier":2,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":32,"verified_fuzzy":43},"total_outbound_references":78},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 78 of 78 outbound references and 2 inbound Pith citation observations for arXiv:2507.01955."}