{"as_of":"2026-08-08T08:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:505a756731867fd07e5b8b541086147d6884ecfcf6f2477d214d9dc8e98fa32c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:45:55.390063Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T15:09:55.285599Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:0b8296dea9c61baac2b6bfff374668090c4da025c73e42c61f7b2629c449d5a8","observation_id":"5c953bd2-e379-4792-a718-1b9d1fd8a12d","resolution":{"observed_at":"2026-05-11T01:19:59.969643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-07T13:45:55.390063Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21061","last_updated":"2025-05-27T11:47:28Z","snapshot_observed_at":"2026-08-07T16:19:17.012074Z","submitted_at":"2025-05-27T11:47:28Z","title":"LPOI: Listwise Preference Optimization for Vision Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T13:45:55.390063Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2505.21061"},"observation_digest":"sha256:3e826f3c64c6596cdda806995599537caa14750e3a7fed2d529d7dc105f3af43","observation_id":"1ebdb2ac-b14b-4722-991f-169e278f991f","resolution":{"observed_at":"2026-08-07T13:45:55.390063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-07T13:14:09.279344Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22396","last_updated":"2025-05-28T14:24:02Z","snapshot_observed_at":"2026-08-07T13:05:45.295882Z","submitted_at":"2025-05-28T14:24:02Z","title":"Zooming from Context to Cue: Hierarchical Preference Optimization for Multi-Image MLLMs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T13:14:09.279344Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2505.22396"},"observation_digest":"sha256:d790b99b9cd050484be8708786f6316c02149ba64c1087a731a7a6bc911fe26b","observation_id":"8038ddba-2872-40b6-9f42-843263afc2fe","resolution":{"observed_at":"2026-08-07T13:14:09.279344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-07T11:08:42.177990Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04277","last_updated":"2025-06-04T02:07:40Z","snapshot_observed_at":"2026-08-07T10:59:34.770632Z","submitted_at":"2025-06-04T02:07:40Z","title":"RSVP: Reasoning Segmentation via Visual Prompting and Multi-modal Chain-of-Thought","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T11:08:42.177990Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2506.04277"},"observation_digest":"sha256:806b7836171c4a91ebeb02ec201b2684d1d4103ebf72f297f480c7a5d6c70f4b","observation_id":"8aa6eeb3-2245-4d99-8f1c-4cef689c819a","resolution":{"observed_at":"2026-08-07T11:08:42.177990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-07T10:28:11.715662Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05302","last_updated":"2025-06-05T17:51:39Z","snapshot_observed_at":"2026-08-07T10:19:12.341695Z","submitted_at":"2025-06-05T17:51:39Z","title":"Perceive Anything: Recognize, Explain, Caption, and Segment Anything in Images and Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:11.715662Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2506.05302"},"observation_digest":"sha256:9c583bf2b2785d6298d2715a2b3870ff54eedc6a47fffb07be1cb3a796e207af","observation_id":"734693f4-cdbe-494d-a851-2c3d8c130a24","resolution":{"observed_at":"2026-08-07T10:28:11.715662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-06T21:27:40.583636Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.24102","last_updated":"2025-06-30T17:51:25Z","snapshot_observed_at":"2026-08-07T09:15:18.123904Z","submitted_at":"2025-06-30T17:51:25Z","title":"DenseWorld-1M: Towards Detailed Dense Grounded Caption in the Real World","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T21:27:40.583636Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2506.24102"},"observation_digest":"sha256:d27855dc2dd5e519821edc60941033a7ed26cb79044f8c3942d930f5da161263","observation_id":"88536864-af88-40b0-8df6-7f58519084cf","resolution":{"observed_at":"2026-08-06T21:27:40.583636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-05T23:35:21.467830Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.05053","last_updated":"2025-08-07T06:10:15Z","snapshot_observed_at":"2026-08-08T05:00:22.796967Z","submitted_at":"2025-08-07T06:10:15Z","title":"Finding Needles in Images: Can Multimodal LLMs Locate Fine Details?","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T23:35:21.467830Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2508.05053"},"observation_digest":"sha256:5808d54eb8f747b79ceaf94bff8a78185a3dfd0b2452284895f8866ba5cf0a97","observation_id":"13065974-bb15-4c14-9943-db6377dce38b","resolution":{"observed_at":"2026-08-05T23:35:21.467830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-05T18:48:03.409499Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXivpreprintarXiv:2403.20271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.14157","last_updated":"2025-08-19T18:00:00Z","snapshot_observed_at":"2026-08-07T00:09:25.806241Z","submitted_at":"2025-08-19T18:00:00Z","title":"The high-speed X-ray camera on AXIS: design and performance updates","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T18:48:03.409499Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2508.14157"},"observation_digest":"sha256:cbcc3428b6f4cfe6a6282b4f7681b28cac5101dcde35ce2e2ff2ab8ee371ef9b","observation_id":"3f185046-efe2-4236-9ccb-86527044b2be","resolution":{"observed_at":"2026-08-05T18:48:03.409499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-08-05T14:01:17.690922Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21809","last_updated":"2025-08-29T17:43:58Z","snapshot_observed_at":"2026-08-05T14:01:14.646681Z","submitted_at":"2025-08-29T17:43:58Z","title":"VoCap: Video Object Captioning and Segmentation from Any Prompt","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T14:01:17.690922Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2508.21809"},"observation_digest":"sha256:f606d01f1b79e4c53fe93914803e6c13a60c996828ac49fb4412fb60109594e1","observation_id":"b819030b-abb2-474b-a43c-1fdb5303f7b7","resolution":{"observed_at":"2026-08-05T14:01:17.690922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2511.18957","last_updated":"2026-04-13T02:31:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-24T10:19:56Z","title":"Eevee: Towards Close-up High-resolution Video-based Virtual Try-on","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-17T06:34:45.045267Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2511.18957"},"observation_digest":"sha256:c1aef6340acb2ae07442596fc16397eb92e672899f20165cc35eb6ae37bab934","observation_id":"3eb49cf9-3455-43e8-8fe9-ea4402bdc796","resolution":{"observed_at":"2026-05-17T06:39:10.444767Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2512.07348","last_updated":"2026-04-28T10:02:14Z","snapshot_observed_at":"2026-08-06T12:32:11.849043Z","submitted_at":"2025-12-08T09:40:11Z","title":"MICo-150K: A Comprehensive Dataset Advancing Multi-Image Composition","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-17T00:20:58.483350Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2512.07348"},"observation_digest":"sha256:7e187339179f1bc977d13c47403db61d83930a056613bc3c1a872340e8efaa6e","observation_id":"a8333f38-25bb-48dd-b407-e8f5f10b7d6e","resolution":{"observed_at":"2026-05-17T00:21:23.357910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2512.09299","last_updated":"2026-04-06T13:16:33Z","snapshot_observed_at":"2026-07-06T22:38:35.387503Z","submitted_at":"2025-12-10T03:57:29Z","title":"VABench: A Comprehensive Benchmark for Audio-Video Generation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T00:03:45.576961Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2512.09299"},"observation_digest":"sha256:703ffebfcdd3e3ee10fbda396f1653e7587e9e690dad8ea0c4762887cd0a4a32","observation_id":"84cc4755-f6aa-404c-a487-402e84ec102e","resolution":{"observed_at":"2026-05-17T00:08:43.880478Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2602.03151","last_updated":"2026-04-06T04:51:44Z","snapshot_observed_at":"2026-07-06T22:44:18.875334Z","submitted_at":"2026-02-03T06:06:35Z","title":"Enhancing Foundation VLM Robustness to Missing Modality: Scalable Diffusion for Bi-directional Feature Restoration","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-16T08:33:49.841678Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2602.03151"},"observation_digest":"sha256:51ac5e22880d8baf91f0f989a641a850818555f872d6440cd7028507442e071e","observation_id":"099aca5f-e86d-4736-b73d-a26ef6ccb138","resolution":{"observed_at":"2026-05-16T08:37:37.247123Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-13T09:40:35.631188Z","title":"Draw-and- understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.04834","last_updated":"2026-06-30T15:42:53Z","snapshot_observed_at":"2026-07-13T09:40:35.215938Z","submitted_at":"2026-04-06T16:35:57Z","title":"E-VLA: Event-Augmented Vision-Language-Action Model for Dark and Blurred Scenes","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-13T09:40:35.631188Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2604.04834"},"observation_digest":"sha256:bc406b23bdec7ee855789cc9cc49a03b1a8270768b5cce330c207d9b5eac7b15","observation_id":"bc407873-50c7-4d54-a77b-68db0bdaf033","resolution":{"observed_at":"2026-07-13T09:40:35.631188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2604.04838","last_updated":"2026-04-07T10:14:44Z","snapshot_observed_at":"2026-07-06T22:53:41.734429Z","submitted_at":"2026-04-06T16:41:19Z","title":"Less Detail, Better Answers: Degradation-Driven Prompting for VQA","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T20:17:01.867903Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2604.04838"},"observation_digest":"sha256:84ea6dd4a0823c7df8ea077270a381dc8b936e088e1d8b64222cd07f785c886a","observation_id":"08b66f69-779f-4a4c-84da-280aaae95099","resolution":{"observed_at":"2026-05-10T22:05:47.909302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2604.11789","last_updated":"2026-04-20T14:38:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-13T17:55:02Z","title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-10T15:35:37.095627Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2604.11789"},"observation_digest":"sha256:84ad4bfc23cc7c22b9297931842994234409cec8ecdb68ca2ddf5aed9afb012b","observation_id":"831e340b-8378-43ec-b7c0-8f747c970060","resolution":{"observed_at":"2026-05-11T10:11:11.116785Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2605.16903","last_updated":"2026-05-16T09:28:46Z","snapshot_observed_at":"2026-07-06T23:27:57.805018Z","submitted_at":"2026-05-16T09:28:46Z","title":"WOW-Seg: A Word-free Open World Segmentation Model","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T21:23:14.311122Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2605.16903"},"observation_digest":"sha256:21275be1a8b9735961c8b8be25502b6436fb3a967644bdcfa9553cc6411ceeb5","observation_id":"b2302e69-4835-4f13-a134-b540e1896a1c","resolution":{"observed_at":"2026-05-19T21:27:48.040652Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2605.27365","last_updated":"2026-05-27T02:30:49Z","snapshot_observed_at":"2026-08-06T02:59:09.668446Z","submitted_at":"2026-05-26T17:59:12Z","title":"LocateAnything: Fast and High-Quality Vision-Language Grounding with Parallel Box Decoding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T17:52:59.346637Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2605.27365"},"observation_digest":"sha256:53596e439e1a0aac12b950996181d253b365ba8969a3f20c2d83f80b59f266cb","observation_id":"ea1e2af6-8aae-4be1-b89c-5dd4ddf80e68","resolution":{"observed_at":"2026-06-29T17:53:46.750455Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":"2403.20271","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-04T15:09:55.285599Z","title":"Draw-and-understand: Leveraging visual prompts to enable mllms to comprehend what you want.arXiv preprint arXiv:2403.20271","venue":null,"work_id":"94e56051-0605-49ff-a17f-e4dd5644593e","year":2024},"citing_paper":{"arxiv_id":"2606.26196","last_updated":"2026-06-24T15:20:32Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T15:20:32Z","title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-26T01:50:54.242508Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2606.26196"},"observation_digest":"sha256:12599335b47f8d681eaa7f0636a9e317c245710cbcd6ab61b5eb5d8293f33e7f","observation_id":"412f60c4-fc1f-48c4-9595-47af92b3b169","resolution":{"observed_at":"2026-07-04T15:09:55.287671Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20271","snapshot_observed_at":"2026-07-14T04:38:05.237334Z","title":"arXiv preprint arXiv:2403.20271 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11581","last_updated":"2026-07-13T14:00:57Z","snapshot_observed_at":"2026-08-06T16:55:31.542502Z","submitted_at":"2026-07-13T14:00:57Z","title":"Actor as Its Own Critic: Unifying Region Understanding and Localization via CycleGRPO","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-14T04:38:05.237334Z"},"links":{"cited_paper":"/paper/2403.20271","citing_paper":"/paper/2607.11581"},"observation_digest":"sha256:3380e652460813324ba4ee24d0f9f9220d2cc4b766642e7680a9638f490a51a4","observation_id":"9ceaa00c-f8bb-40b8-93d5-11c989a28ae9","resolution":{"observed_at":"2026-07-14T04:38:05.237334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.20271/citation-record","integrity":"/paper/2403.20271/integrity","json":"/paper/2403.20271/citation-record.json","paper":"/paper/2403.20271"},"outbound":[],"paper":{"arxiv_id":"2403.20271","last_updated":"2025-02-22T14:02:39Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T00:53:06.955281Z","submitted_at":"2024-03-29T16:26:20Z","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2403.20271."}