{"as_of":"2026-08-21T14:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e2ca415a4ec44c2dbd44ddd033864d8061ca63101a1c49d06b348bef9c544f01","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:47:21.063096Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T08:49:42.520344Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2404.13076","last_updated":"2024-04-15T16:49:59Z","snapshot_observed_at":"2026-08-18T02:44:37.072871Z","submitted_at":"2024-04-15T16:49:59Z","title":"LLM Evaluators Recognize and Favor Their Own Generations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T18:44:28.766639Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2404.13076"},"observation_digest":"sha256:8e3786a382edf2a13c6367cf859b1da042f7c88cffafce39a629a76b2866f6be","observation_id":"645e2664-f561-446e-b5f4-cb4fb8d7985d","resolution":{"observed_at":"2026-05-22T18:44:28.800274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2405.19088","last_updated":"2026-04-15T02:26:56Z","snapshot_observed_at":"2026-07-06T18:21:57.091728Z","submitted_at":"2024-05-29T13:51:43Z","title":"Cracking the Code of Juxtaposition: Can AI Models Understand the Humorous Contradictions","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-24T00:52:52.056076Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2405.19088"},"observation_digest":"sha256:8ae5bd03d81b905f652c5d512dbf81cabe58f1fb87061a1e070e0e731544283d","observation_id":"97bf3576-2483-41ec-bf3b-e25af6f00177","resolution":{"observed_at":"2026-05-24T00:53:40.644813Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2406.03520","last_updated":"2024-10-03T17:24:40Z","snapshot_observed_at":"2026-08-16T12:54:37.000371Z","submitted_at":"2024-06-05T17:53:55Z","title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T11:34:37.599691Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2406.03520"},"observation_digest":"sha256:589e24d37fb1e3db0d63d61f66519c7909a9d17e0eba98fdf57525e485ae2c95","observation_id":"6324c99f-25a7-4c9b-b78d-73f539624710","resolution":{"observed_at":"2026-05-20T11:34:37.649327Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-08-13T00:35:03.634903Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:086129237bfef43779218b928c7c9c9f49623dfd84090459a70606e76d85c2ca","observation_id":"151a6662-a5c1-4d3e-b2f4-2db3112de5ee","resolution":{"observed_at":"2026-05-16T07:59:32.821120Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2410.14702","last_updated":"2026-05-10T19:30:43Z","snapshot_observed_at":"2026-08-19T19:02:34.472298Z","submitted_at":"2024-10-06T20:35:41Z","title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-23T20:03:38.336841Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2410.14702"},"observation_digest":"sha256:1d6f410b7a341823656a4132a46e958e4b4895b05128c521a81f5d2126362681","observation_id":"6c52f880-66dc-44bc-b6d0-94ef03a84f3a","resolution":{"observed_at":"2026-05-23T20:05:47.842622Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-08-12T14:31:36.808308Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15296","last_updated":"2024-12-08T04:24:31Z","snapshot_observed_at":"2026-08-18T21:58:09.907460Z","submitted_at":"2024-11-22T18:59:54Z","title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T14:31:36.808308Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2411.15296"},"observation_digest":"sha256:70416978660c624d78008c56228dff78da64b1d98eac46755a9eb072ad46a47b","observation_id":"448baf20-ff7e-400e-9967-8ca766215267","resolution":{"observed_at":"2026-08-12T14:31:36.808308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-08-11T21:29:26.274862Z","title":"Visit-bench: A benchmark for vision- language instruction following inspired by real-world use","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.04378","last_updated":"2025-05-08T19:16:29Z","snapshot_observed_at":"2026-08-14T18:05:39.883853Z","submitted_at":"2024-12-05T17:54:27Z","title":"VladVA: Discriminative Fine-tuning of LVLMs","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T21:29:26.274862Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2412.04378"},"observation_digest":"sha256:52eb7272e9b41db89527b6df8c5fbb20ef6cc1b78bbee191c3051d26a3b4ce61","observation_id":"9d71215c-51e5-4115-a899-f61f15b75b00","resolution":{"observed_at":"2026-08-11T21:29:26.274862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-08-07T18:23:49.564656Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.10391","last_updated":"2025-02-14T18:59:51Z","snapshot_observed_at":"2026-08-21T12:43:24.932114Z","submitted_at":"2025-02-14T18:59:51Z","title":"MM-RLHF: The Next Step Forward in Multimodal LLM Alignment","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T18:23:49.564656Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2502.10391"},"observation_digest":"sha256:eb7a1c931895fc827acc3460ffdc1b9081d3a307d81b3e78dba6cfd2987314c0","observation_id":"b6300708-95dd-4364-98a6-de3baf853e30","resolution":{"observed_at":"2026-08-07T18:23:49.564656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2503.23137","last_updated":"2026-04-15T02:38:45Z","snapshot_observed_at":"2026-08-12T16:01:25.076287Z","submitted_at":"2025-03-29T16:08:51Z","title":"When 'YES' Meets 'BUT': Can Large Models Comprehend Contradictory Humor Through Comparative Reasoning?","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T22:38:35.969273Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2503.23137"},"observation_digest":"sha256:1d469cb9179d992fe9456fae62f8e4d4d2eb63ddc1dcb75cf225b38d0dc47a6d","observation_id":"6b1111fa-762f-4928-9212-46b65acf39e4","resolution":{"observed_at":"2026-05-22T22:42:13.662482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-08-07T11:32:05.026455Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use.arXiv preprint arXiv:2308.06595, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02308","last_updated":"2025-06-06T21:40:20Z","snapshot_observed_at":"2026-08-15T06:45:25.492982Z","submitted_at":"2025-06-02T22:55:23Z","title":"MINT: Multimodal Instruction Tuning with Multimodal Interaction Grouping","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:05.026455Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2506.02308"},"observation_digest":"sha256:aae9667619bfdba0ccac459a14d4d2dcb95a215fce51f66d019d64fa57860a74","observation_id":"cd280595-afce-46ff-96b7-565d47a44a47","resolution":{"observed_at":"2026-08-07T11:32:05.026455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-08-05T10:51:44.475907Z","title":"Yuanpu Cao, Tianrong Zhang, Bochuan Cao, Ziyi Yin, Lu Lin, Fenglong Ma, and Jinghui Chen","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.03647","last_updated":"2026-06-22T18:38:14Z","snapshot_observed_at":"2026-08-19T21:41:28.995926Z","submitted_at":"2025-09-03T18:52:55Z","title":"Breaking the Mirror: Activation-Based Mitigation of Self-Preference in LLM Evaluators","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-05T10:51:44.475907Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2509.03647"},"observation_digest":"sha256:560ea53247413167ece5ecb92e431739b7a0b4e1f084f95ed04adc79147bf954","observation_id":"8a537809-b3ad-4d46-a0ed-39b94ad498ba","resolution":{"observed_at":"2026-08-05T10:51:44.475907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-08-05T10:16:51.066071Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.04292","last_updated":"2025-09-04T15:03:02Z","snapshot_observed_at":"2026-08-11T06:56:43.869188Z","submitted_at":"2025-09-04T15:03:02Z","title":"Inverse IFEval: Can LLMs Unlearn Stubborn Training Conventions to Follow Real Instructions?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T10:16:51.066071Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2509.04292"},"observation_digest":"sha256:9ba9a8277339e32f0ae5caa2b750a2455f1efd8840b7a8a09f30d57481c87ffd","observation_id":"ba44526c-1df7-4e12-9fab-bb70b0da45ab","resolution":{"observed_at":"2026-08-05T10:16:51.066071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-08-15T15:47:21.063096Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.25339","last_updated":"2026-05-24T10:24:38Z","snapshot_observed_at":"2026-08-16T13:44:23.472388Z","submitted_at":"2025-09-29T18:00:25Z","title":"VisualOverload: Probing Visual Understanding of VLMs in Really Dense Scenes","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-15T15:47:21.063096Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2509.25339"},"observation_digest":"sha256:1c125d30dae1fcb4922e5f04402b5338ef384633dcc61501a4e9cc449f19fbb7","observation_id":"9b10dda8-45ad-4bc0-8edf-8be54046ef0e","resolution":{"observed_at":"2026-08-15T15:47:21.063096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2604.18803","last_updated":"2026-04-25T21:48:41Z","snapshot_observed_at":"2026-08-12T17:43:47.112004Z","submitted_at":"2026-04-20T20:21:27Z","title":"LLM-as-Judge Framework for Evaluating Tone-Induced Hallucination in Vision-Language Models","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T05:11:01.039309Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2604.18803"},"observation_digest":"sha256:e20b79f8bb246d432db691ade0fa96b8d6493bd0cee21141242cb08cfadccc14","observation_id":"07d9bdea-0282-4803-a703-bddc4dd1e19f","resolution":{"observed_at":"2026-05-10T09:38:43.153061Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2606.06217","last_updated":"2026-06-04T14:31:11Z","snapshot_observed_at":"2026-08-16T04:23:20.384614Z","submitted_at":"2026-06-04T14:31:11Z","title":"DisasterBench: A Multimodal Benchmark for UAV-Based Disaster Response in Complex Environments","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-28T02:36:11.721143Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2606.06217"},"observation_digest":"sha256:7cff97707a41e560b2f899b6ee7e488ab3fc429a97d2a5665d68cfcef84a587d","observation_id":"bdf34ac1-4e87-4854-b000-b80797822dcd","resolution":{"observed_at":"2026-07-02T11:56:55.937243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2606.22476","last_updated":"2026-06-21T12:35:43Z","snapshot_observed_at":"2026-08-17T01:01:06.764669Z","submitted_at":"2026-06-21T12:35:43Z","title":"CVSBench: A Comprehensive Benchmark for Cross-view Spatial Reasoning and Dreaming","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T10:54:49.774490Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2606.22476"},"observation_digest":"sha256:bd88ee0dbbe6abacd9eb3d3e23ee7faf99dbd771c0ae4a4f65324448a0f89376","observation_id":"07333a90-e63f-471b-9bdb-05a47d8ed0c3","resolution":{"observed_at":"2026-07-04T08:49:42.521679Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","version":4},"cited_work":{"arxiv_id":"2308.06595","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.06595","snapshot_observed_at":"2026-07-04T08:49:42.520344Z","title":"Visit-bench: A benchmark for vision-language instruction following inspired by real-world use","venue":null,"work_id":"43af94c6-d4f4-4441-a85d-3c252dc55850","year":2023},"citing_paper":{"arxiv_id":"2606.30556","last_updated":"2026-06-29T16:51:31Z","snapshot_observed_at":"2026-08-07T13:43:00.994295Z","submitted_at":"2026-06-29T16:51:31Z","title":"Poller: Are LLMs Suitable for Evaluating the Poetry Understanding Task?","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-30T05:59:58.183264Z"},"links":{"cited_paper":"/paper/2308.06595","citing_paper":"/paper/2606.30556"},"observation_digest":"sha256:b0bdf56c5d374efc449afa801c541ce2ca704a1ecca5dfa9199f1ca1df0e597d","observation_id":"c6619a0e-fa7a-4650-b7a9-652ab7d0f249","resolution":{"observed_at":"2026-06-30T06:04:21.562417Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2308.06595/citation-record","integrity":"/paper/2308.06595/integrity","json":"/paper/2308.06595/citation-record.json","paper":"/paper/2308.06595"},"outbound":[],"paper":{"arxiv_id":"2308.06595","last_updated":"2023-12-26T15:57:47Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-18T16:51:00.679875Z","submitted_at":"2023-08-12T15:27:51Z","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2308.06595."}