{"as_of":"2026-08-11T07:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9e66509b9c5681610c4a40956a76c2577da69f35d3d69103f8ddd1af09137555","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:08:46.326666Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:34:58.344234Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-14T00:18:29.517494Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-07T14:34:58.344234Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18434","last_updated":"2025-05-24T00:02:48Z","snapshot_observed_at":"2026-08-10T11:46:31.778304Z","submitted_at":"2025-05-24T00:02:48Z","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T14:34:58.344234Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2505.18434"},"observation_digest":"sha256:8149bc0a878aa9b7354cc887c18abf6a7a7771849aa91622aecc09ef66e1447f","observation_id":"6a51e034-03ea-4726-a7a1-89714e6234a0","resolution":{"observed_at":"2026-08-07T14:34:58.344234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-07T13:01:09.271049Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22946","last_updated":"2025-05-28T23:58:37Z","snapshot_observed_at":"2026-08-09T17:04:53.831723Z","submitted_at":"2025-05-28T23:58:37Z","title":"NegVQA: Can Vision Language Models Understand Negation?","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T13:01:09.271049Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2505.22946"},"observation_digest":"sha256:729db4c0d5ef6ab8c0e9ec1a62e9002bac7177f22844a80ec0f444991915fded","observation_id":"93ca8fa6-82d7-43c1-853c-34dbd8976c31","resolution":{"observed_at":"2026-08-07T13:01:09.271049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-04T20:28:21.003431Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09732","last_updated":"2025-09-10T13:08:03Z","snapshot_observed_at":"2026-08-11T03:15:31.994363Z","submitted_at":"2025-09-10T13:08:03Z","title":"Decomposing Visual Classification: Assessing Tree-Based Reasoning in VLMs","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-04T20:28:21.003431Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2509.09732"},"observation_digest":"sha256:6836d12e21159deb067558925dfdc52918f8a14c4bc7efdb83638cd937ac134b","observation_id":"af5a841d-b63e-4378-9185-5f06f40dddbf","resolution":{"observed_at":"2026-08-04T20:28:21.003431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2604.18942","last_updated":"2026-04-21T00:32:18Z","snapshot_observed_at":"2026-07-06T23:05:39.724237Z","submitted_at":"2026-04-21T00:32:18Z","title":"Disparities In Negation Understanding Across Languages In Vision-Language Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-10T03:31:50.234201Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2604.18942"},"observation_digest":"sha256:a0969c02c503a31f9f9ec75b4dffc2e558bd57e60837879b95c29927e3a12e49","observation_id":"65823167-c3b5-467e-9557-9740e1845312","resolution":{"observed_at":"2026-05-11T12:31:03.990536Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2604.22773","last_updated":"2026-03-31T03:05:26Z","snapshot_observed_at":"2026-07-06T23:09:05.682387Z","submitted_at":"2026-03-31T03:05:26Z","title":"Trace Mutation in Human-LLM Dialogue: The Transcript as Forensic and Mitigation Surface","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-14T00:18:14.010081Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2604.22773"},"observation_digest":"sha256:ddb961e2994b133f602e52277f0d2ce345c2601e7b05a71f7e751af672ce6076","observation_id":"8f28e7c4-ea6d-489f-935b-6c6cc2a6296d","resolution":{"observed_at":"2026-05-14T00:18:29.524884Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2605.05810","last_updated":"2026-05-07T07:46:17Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T07:46:17Z","title":"CXR-ContraBench: Benchmarking Negated-Option Attraction in Medical VLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T14:49:53.357083Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2605.05810"},"observation_digest":"sha256:9543e82c1d70b0c678e85cca50defbd3ff93372f986786a4a395c282d9ca4a45","observation_id":"2471455d-3f34-45b8-8239-b49ecff50ba2","resolution":{"observed_at":"2026-05-11T18:41:09.727332Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":"2501.09425","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","venue":null,"work_id":"c69c2a80-b68f-48b4-aed9-61f51c531fe2","year":2025},"citing_paper":{"arxiv_id":"2605.06815","last_updated":"2026-05-07T18:16:49Z","snapshot_observed_at":"2026-08-04T01:42:42.718743Z","submitted_at":"2026-05-07T18:16:49Z","title":"Uneven Evolution of Cognition Across Generations of Generative AI Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-11T01:00:59.086900Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2605.06815"},"observation_digest":"sha256:d3887d08ff7d00741e994d57e6c1b634de9231fedb790636e4ed588b2c417ea9","observation_id":"10e8c895-0807-4b6d-a917-130295408816","resolution":{"observed_at":"2026-05-11T04:50:58.108557Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09425","snapshot_observed_at":"2026-08-01T17:16:49.314099Z","title":"Available: https://arxiv.org/abs/2501.09425","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17712","last_updated":"2026-07-20T09:08:51Z","snapshot_observed_at":"2026-08-03T12:43:11.710062Z","submitted_at":"2026-07-20T09:08:51Z","title":"Learning to Detect Cross-Modal Negation: An Analysis of Latent Representations and an Attention-Based Solution","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T17:16:49.314099Z"},"links":{"cited_paper":"/paper/2501.09425","citing_paper":"/paper/2607.17712"},"observation_digest":"sha256:1ca284e0f3619f504366af6b25588a293fb6628d0d23d1e7dccb3f45f2e5b77f","observation_id":"64c36674-3a95-4912-9f7c-3014d8fc0d3a","resolution":{"observed_at":"2026-08-01T17:16:49.314099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.09425/citation-record","integrity":"/paper/2501.09425/integrity","json":"/paper/2501.09425/citation-record.json","paper":"/paper/2501.09425"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.563891Z","title":"Aug- mented reality meets computer vision: Efficient data gen- eration for urban driving scenes","venue":null,"work_id":"bbd8caea-a74a-4a2b-8162-c4ed076d11d6","year":2018},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.008226Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:19b468b0cfea4b3e4e6f594c0f6cb0a396665d08e56ee36bc9a9d909b69a839d","observation_id":"876019cf-4e11-484d-93b7-b82286f09e37","resolution":{"observed_at":"2026-08-10T20:08:47.570497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.541877Z","title":"Effective conditioned and composed im- age retrieval combining clip-based features","venue":null,"work_id":"c1e8940f-4bde-4794-928a-0464d362e5e1","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.014749Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:9980a097afa4dd9c7f370fcffa2c59bd87efa40030fed787b23a6377715cc804","observation_id":"76626a12-b8c4-4540-861e-c24ddca9336b","resolution":{"observed_at":"2026-08-10T20:08:47.548229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.511347Z","title":"FitCLIP: Refining large- scale pretrained image-text models for zero-shot video un- derstanding tasks","venue":null,"work_id":"5dc3b93e-f045-4352-af50-c1e86fe03be9","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.020123Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:43da83ad24cd0e86aa95c436f3eefe23335fec45be210e9622d6ec3a4067c91d","observation_id":"5d08414e-b6b6-492a-ac86-dbac81d7a1fe","resolution":{"observed_at":"2026-08-10T20:08:47.517244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.031905Z","title":"Conceptual 12M: Pushing web-scale image-text pre-training to recognize long-tail visual concepts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.031905Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:b3235bf29b730bba120fbb817b3ac68e98081ab08dd1e970debad589902dde15","observation_id":"adc40a9b-4ad3-44b7-96a0-b98ec2b7ff5d","resolution":{"observed_at":"2026-08-10T20:08:46.031905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.447175Z","title":"Learning semantic segmentation from synthetic data: A geo- metrically guided input-output adaptation approach","venue":null,"work_id":"a345b268-77a8-4cff-ab65-46842644f4d9","year":2019},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.037424Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:06f5ee1773c1873c42584640562d5ae1d38dcfaf568761de8275bacb0a557f81","observation_id":"19db9dd0-2961-4ed7-ad64-5bae5646be84","resolution":{"observed_at":"2026-08-10T20:08:47.453953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-10T20:08:46.042836Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.042836Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:15a0a0c85327708e90c672349dea7b71f841f6bcda8fdc8142f4175a518b488f","observation_id":"3e0cc5c6-38ab-4ba6-855b-8a80c4030f3d","resolution":{"observed_at":"2026-08-10T20:08:46.042836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.421837Z","title":"The pascal visual object classes (voc) challenge","venue":null,"work_id":"a7de39c3-229b-49d1-9e38-d9af2048a2c8","year":2010},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.048534Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:18ef5ca91e88d4ca24e0d91b3c9cfa624af3d7014b1e984edc5fd2ccd1b90c8f","observation_id":"97d27b26-b7c2-45d5-ab5c-dbe351a3529a","resolution":{"observed_at":"2026-08-10T20:08:47.429941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14402","last_updated":"2024-11-21T18:31:25Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:31:25Z","title":"Multimodal Autoregressive Pre-training of Large Vision Encoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14402","snapshot_observed_at":"2026-08-10T20:08:46.054919Z","title":"Mul- timodal autoregressive pre-training of large vision encoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.054919Z"},"links":{"cited_paper":"/paper/2411.14402","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:7c77bc4a60401558073fb55135eef20fe783dff6b6c7e5fc16486690af0a66b3","observation_id":"11613278-2c26-4b96-9ae0-c64f4c515b2a","resolution":{"observed_at":"2026-08-10T20:08:46.054919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.395763Z","title":"Datacomp: In search of the next generation of multimodal datasets","venue":null,"work_id":"d5915288-4623-4e5e-ab48-f707c03d31d4","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.061090Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:80f5f84d657f53515bae81b0841d6f9ebb91a951578efb32d1a991e6db1eaec5","observation_id":"43de77e5-ed52-44ba-8ed4-68d1bedef7ab","resolution":{"observed_at":"2026-08-10T20:08:47.402544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.373394Z","title":"This is not a dataset: A large negation benchmark to challenge large language mod- els","venue":null,"work_id":"3ccc5764-ddd9-4ab6-bb4e-6dfcd49a9c02","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.076557Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:72e4737892ea802f6e791dee93c4d4d7acbeb6b52c7fe6c6a1f76ac6111bd3f2","observation_id":"e1117c67-0df2-4e42-a899-fb9d4b5c7bcd","resolution":{"observed_at":"2026-08-10T20:08:47.382504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.349563Z","title":"Shortcut learning in deep neural networks","venue":null,"work_id":"32c3e351-2275-472f-add0-3ba99e279125","year":2020},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.081422Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:b72d0f082d631440c1d63e85ac313cb8dc4065b7cecf736a9aa2e481d442318f","observation_id":"504dc068-fbb3-4026-b1d4-87f3c4e9d131","resolution":{"observed_at":"2026-08-10T20:08:47.355708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01832","last_updated":"2024-07-18T10:21:29Z","snapshot_observed_at":"2026-08-04T15:45:57.991866Z","submitted_at":"2024-02-02T18:59:58Z","title":"SynthCLIP: Are We Ready for a Fully Synthetic CLIP Training?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01832","snapshot_observed_at":"2026-08-10T20:08:46.086165Z","title":"Synthclip: Are we ready for a fully synthetic clip training? arXiv preprint arXiv:2402.01832, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.086165Z"},"links":{"cited_paper":"/paper/2402.01832","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f2fb4155f2e23131e2dfd83e8dff1206147c3415398924a869157fd21ffaa22b","observation_id":"6ad3e1d3-2a2b-4509-be2f-5b1e12a3db5e","resolution":{"observed_at":"2026-08-10T20:08:46.086165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.329550Z","title":null,"venue":null,"work_id":"fd704792-6614-4823-adca-ca5400697b7f","year":1989},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.091030Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:98f8a08dd546c7dfbabca1d188f045dd0422dc7022004deed586b1ca13c1d22e","observation_id":"863bde2d-1a1e-433d-bfcd-09a3e99ecd6e","resolution":{"observed_at":"2026-08-10T20:08:47.335828Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.308277Z","title":"Quilt-1m: One million image-text pairs for histopathology","venue":null,"work_id":"784cb7cf-c802-41e9-b550-7ac75826b6c9","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.096959Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:86d1860e610dee823e1d48bebb58bc60999461d4a1c1ae47a57a5c67191d330e","observation_id":"fd484441-67ad-4796-b2bf-0cc61cb951c7","resolution":{"observed_at":"2026-08-10T20:08:47.314683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.289449Z","title":"Chexpert: A large chest radiograph dataset with uncertainty labels and expert comparison","venue":null,"work_id":"4f59563c-f337-4a6d-b0b7-89b8b8a5172a","year":2019},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.103106Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:2853c8cd34100b59a1841fb620b462bcb09c65e5e4513d197d5ae022359126a3","observation_id":"73531a14-883c-4bee-842c-7c0fc3a33525","resolution":{"observed_at":"2026-08-10T20:08:47.296473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.270647Z","title":"Generative models as a data source for multiview representa- tion learning","venue":null,"work_id":"579643ec-1ba6-42c6-bcd6-3acc9573ee71","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.108782Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f3120e6977d90b52b79dd867ed46bb8333bd7a63cd629f1a187b24c2b48e2434","observation_id":"a0594ace-e525-4638-b328-70feab20beaa","resolution":{"observed_at":"2026-08-10T20:08:47.277193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.251670Z","title":"The power of negation in english: Text, context and relevance","venue":null,"work_id":"b6f4a73b-1178-41cd-8edc-41c20dad13cf","year":1998},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.113421Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:7e596287012146738b492ef932af898350b8f1bd40966c128fae3ddf6191d5d9","observation_id":"6825bbb5-161f-4b27-8c15-85c4acb02932","resolution":{"observed_at":"2026-08-10T20:08:47.257383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.228306Z","title":"Negation in syntax–on the na- ture of functional categories and projections","venue":null,"work_id":"7f096a3a-092a-47d4-80a3-9c93a340da10","year":1990},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.118134Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:e38728484e1c0bd19eeb0895970088a0a2b6f25f1fefd12d9279ac8db449fb3c","observation_id":"b40567b7-2ebc-44d1-a989-c1c016b6515d","resolution":{"observed_at":"2026-08-10T20:08:47.234939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.208449Z","title":"Naturalbench: Evalu- ating vision-language models on natural adversarial samples","venue":null,"work_id":"f659f5a2-691c-4571-a68b-b61bf4689a5d","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.122815Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:dc3a81ff66dced3dd9e89a31c5c9bdabc29cf97e2044681d279e5974992b6155","observation_id":"e4b8857b-2a24-492a-8b01-221b1f34db22","resolution":{"observed_at":"2026-08-10T20:08:47.214576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.187847Z","title":"Compre- hending and ordering semantics for image captioning","venue":null,"work_id":"53389f50-f88b-4364-8ebc-9daf9aca9c42","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.128486Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:34fda7512b2f68bfe309cd815b4914aa550879c883f0d1a8016d5491f3f96730","observation_id":"51eb013c-e984-4379-8ec6-8ce2bcc29324","resolution":{"observed_at":"2026-08-10T20:08:47.194283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.168069Z","title":"Cross-modal retrieval and semantic re- finement for remote sensing image captioning","venue":null,"work_id":"64ef388b-d927-49f7-8c54-c25ba2e81f5c","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.133410Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:35e1ff232d1dc33a8e84603fb0c7aa2ca861b74453db038adfe5b8231fefdbaa","observation_id":"e2d73bba-3d18-4099-bf07-0007c3bfce36","resolution":{"observed_at":"2026-08-10T20:08:47.174919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.150497Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"7c184874-edbf-42e1-9be5-a02c15bab4ea","year":2014},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.138330Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:befd56ffba895fd8df3e4a8cb3a02e7a27a4a767d5c4980a8d0abb4a5fb5e773","observation_id":"9c371f82-ffe9-4c86-b92b-0595593f0e24","resolution":{"observed_at":"2026-08-10T20:08:47.156422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.127762Z","title":"A visual- language foundation model for computational pathology","venue":null,"work_id":"067b4534-14d3-48bc-be4b-fd5e9e953677","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.143444Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:3f8c2a7221730bb0b133831c51d02b8acc01fe79ae5723c75ea4fe86d2920077","observation_id":"c376ff10-6159-490e-b8f8-24a60b6a017e","resolution":{"observed_at":"2026-08-10T20:08:47.134814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08860","last_updated":"2021-05-08T08:25:57Z","snapshot_observed_at":"2026-08-10T14:43:16.360554Z","submitted_at":"2021-04-18T13:59:50Z","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08860","snapshot_observed_at":"2026-08-10T20:08:46.149164Z","title":"CLIP4Clip: An empirical study of clip for end to end video clip retrieval","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.149164Z"},"links":{"cited_paper":"/paper/2104.08860","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:a829d67048e5533437c179be1e1239f0bb59db1f4fe4a67f6cc14ce35a0c0b56","observation_id":"92c26dc6-4b83-4553-a6ca-f3662ab335d6","resolution":{"observed_at":"2026-08-10T20:08:46.149164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.099166Z","title":"Fine-tuning llama for multi-stage text retrieval","venue":null,"work_id":"4b14f085-b502-44b0-9634-12a17b42a58c","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.154599Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:bea83d2a4f724e0d106af498588135342e11a19bad9984c837235c8833864523","observation_id":"46d0c712-e0cf-4ceb-9a8a-be0410d8928e","resolution":{"observed_at":"2026-08-10T20:08:47.105551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.061659Z","title":"Crepe: Can vision-language foundation models reason compositionally? In CVPR, 2023","venue":null,"work_id":"2a445046-a557-42a3-b494-8d3e525d3362","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.159371Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f80bdd84e61f467095bf54444c8db9c3cb7005c6d79e8ab56af01d437c42043c","observation_id":"e298a188-b5e0-4d97-b981-26f8a39358a5","resolution":{"observed_at":"2026-08-10T20:08:47.077283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.040106Z","title":"Simple open-vocabulary object detection with vi- sion transformers","venue":null,"work_id":"26bf24de-69f0-4c34-ab97-179449e9decb","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.164498Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:1f18a136de2be62654813dee713f804abcbc4aafdb314e902efccde8152e9fa7","observation_id":"367d25f1-9eed-41c0-bc92-6b3d86ddcd67","resolution":{"observed_at":"2026-08-10T20:08:47.046531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.015446Z","title":"Recent advances in processing negation","venue":null,"work_id":"09847457-4797-460c-9b1c-03f193eaf114","year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.169611Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:6813fd94ff249e0b6798a3e7ee08280f06cb4498ae2f830e473c9c7639e66249","observation_id":"7bb530a6-5b4e-447e-983a-dc3ff370dd56","resolution":{"observed_at":"2026-08-10T20:08:47.021702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.982419Z","title":"Effect of negation in sentences on sentiment analy- sis and polarity detection","venue":null,"work_id":"3666c431-3ff9-438b-a613-a51c8a2cef3e","year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.174089Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:d2ec143520d30e0e4d027ac0e5030e0acb5af859a664e2c1155e5e7572712664","observation_id":"98baac9d-107f-4d94-9a8e-fc3102221a5b","resolution":{"observed_at":"2026-08-10T20:08:46.990672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.961213Z","title":"Clip-it! language-guided video summarization","venue":null,"work_id":"235c0f93-d366-45c8-aee9-59de81656a6b","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.178920Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:79936eeb915d275a9914323adeb8fa4708b0eb18963db22e2b6e752b78acbac8","observation_id":"1e9c7f92-9608-4b9a-bbb0-8f78ca2618b7","resolution":{"observed_at":"2026-08-10T20:08:46.967778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.14424","last_updated":"2019-10-31T12:45:40Z","snapshot_observed_at":"2026-08-09T22:20:54.188721Z","submitted_at":"2019-10-31T12:45:40Z","title":"Multi-Stage Document Ranking with BERT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.14424","snapshot_observed_at":"2026-08-10T20:08:46.185183Z","title":"Multi-stage document ranking with bert","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.185183Z"},"links":{"cited_paper":"/paper/1910.14424","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:62c736ec9c1fe83ad63ceb9d1b1fafa5a080fd6337cadabb78dd33fc93c2a4ce","observation_id":"8835517b-641d-4731-9c1e-a04012018817","resolution":{"observed_at":"2026-08-10T20:08:46.185183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.940262Z","title":"Synthesize diagnose and optimize: Towards fine- grained vision-language understanding","venue":null,"work_id":"a86dc088-a867-40df-8d62-2901acf97bff","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.190882Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:d9fbafb29c8dd394566b10854b37d8a15c502d213011eceb3139336dedad0829","observation_id":"5869f8b3-04a3-4e44-a339-5be9332767af","resolution":{"observed_at":"2026-08-10T20:08:46.947397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.920315Z","title":"On guiding vi- sual attention with language specification","venue":null,"work_id":"06b6325d-65aa-41f5-95be-6a950447c8fe","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.198307Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:18df55ec20264d462cd0824e519c06ed4797d03a06e21efd906fc848a30ce48b","observation_id":"3e17222d-e85e-4157-8b69-7f9b531211b2","resolution":{"observed_at":"2026-08-10T20:08:46.926944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.903520Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"142d36a8-8493-46bc-9bd9-319fbc1e426d","year":2021},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.203619Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:6ff19940b39c6e23a5b85f3fbf1091070a254bd5b946db559b9d5fb00fab55d0","observation_id":"877627b6-c6ca-4b13-89d2-df8a6aa91d57","resolution":{"observed_at":"2026-08-10T20:08:46.909588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.886237Z","title":"Denseclip: Language-guided dense prediction with context- aware prompting","venue":null,"work_id":"a51d9a37-74ed-4e30-aa28-874cb69c5419","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.208659Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:4e1c52867bf0db616a3bc0661720f8af7471e2905e84fb249899e027533c0ea8","observation_id":"1a058905-4956-4332-b0bb-e67c043cadab","resolution":{"observed_at":"2026-08-10T20:08:46.891949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.868007Z","title":"Sentence-bert: Sentence embeddings using siamese bert-networks","venue":null,"work_id":"19d659e0-95cb-4396-912e-f03f2db66a08","year":2019},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.214930Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:0278fad365d8300816ae0ea79b97956d27524bd0dd4bd63ef2c656f935b81a5d","observation_id":"b2cf1896-10f9-4896-a3a8-551b340ce5b3","resolution":{"observed_at":"2026-08-10T20:08:46.875485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.850248Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":"2eab8cd7-60be-4627-baf2-7db994f1f4c2","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.221528Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:4ad21e421eae977b0eb781ffb6fc538b9bba449cb9a7339a6bf1fb11413619bc","observation_id":"986b4a43-0592-49e9-b54d-5fdb299a5c7d","resolution":{"observed_at":"2026-08-10T20:08:46.855890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.831533Z","title":"Clip for all things zero-shot sketch-based image retrieval, fine- grained or not","venue":null,"work_id":"17b65315-7408-4575-a0d1-a868bd98e432","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.226461Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:55fd58d9bae2f49781eb10c2e82a0d915be1a14c131e222caddde0ae97f7a558","observation_id":"850b4137-6f56-4bb0-8303-dfb1b0199aa7","resolution":{"observed_at":"2026-08-10T20:08:46.837669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.792596Z","title":"LAION-5b: An open large-scale dataset for train- ing next generation image-text models","venue":null,"work_id":"a84f676d-7da3-4d19-a565-3c7d122f28b5","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.231797Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:9d2563aa5f0642d61707b2bd94bf0c60266417f4dbb099513e8759ff1602f067","observation_id":"1a841611-2507-45f8-b381-b5259404120b","resolution":{"observed_at":"2026-08-10T20:08:46.819673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.773321Z","title":"How much can clip benefit vision-and-language tasks? In International Conference on Learning Representa- tions","venue":null,"work_id":"9de1d9fc-9a25-4513-aba7-c758a00f5235","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.238053Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:22b47d227dd7a69da8bfe5e2bd0dc66b18e5de4548fde361cda9ac09561dc658","observation_id":"b1117bc4-936e-443f-b7f3-fcd2367020e7","resolution":{"observed_at":"2026-08-10T20:08:46.779307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.751904Z","title":"Proposalclip: Unsupervised open-category object pro- posal generation via exploiting clip cues","venue":null,"work_id":"23d167a7-8ae0-47d6-825e-e43708af1178","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.244630Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:1d7ad7a5fe67a3bddcaad162dec371182d9da62425eba101b1ec182b066baf72","observation_id":"89029596-1604-4d2e-8baa-02b6076886a1","resolution":{"observed_at":"2026-08-10T20:08:46.759276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.732951Z","title":"Cliport: What and where pathways for robotic manipulation","venue":null,"work_id":"caaf6ed3-5c36-4b2c-9a0a-41bad9752c61","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.250104Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:c09ce6a6853032a55a166b32250c72e2b80932197dc5de180a611598aeab651e","observation_id":"15f7bafa-b741-49ac-964c-95d392e8eba8","resolution":{"observed_at":"2026-08-10T20:08:46.739522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.20312","last_updated":"2024-03-29T17:33:42Z","snapshot_observed_at":"2026-08-10T14:28:10.363083Z","submitted_at":"2024-03-29T17:33:42Z","title":"Learn \"No\" to Say \"Yes\" Better: Improving Vision-Language Models via Negations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.20312","snapshot_observed_at":"2026-08-10T20:08:46.254987Z","title":"Learn” no” to say” yes” bet- ter: Improving vision-language models via negations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.254987Z"},"links":{"cited_paper":"/paper/2403.20312","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:e83aeebfffd57ec2e9fe6e57e6032d8f155d3243ca97df05d52b664ca60d558a","observation_id":"6b9ddda6-4d36-47c2-8059-4939e8fb9485","resolution":{"observed_at":"2026-08-10T20:08:46.254987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.689611Z","title":"Stablerep: Synthetic images from text-to- image models make strong visual representation learners","venue":null,"work_id":"29ac4242-e750-4e52-968b-2b36fb0b0355","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.260007Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:37d4253782abc6e78b732af1432df54843c80a557c09dae83ce29386382ed3d3","observation_id":"e6d9dcd2-4121-4365-b480-4979e54b9d75","resolution":{"observed_at":"2026-08-10T20:08:46.707901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.664764Z","title":"Learning vision from mod- els rivals learning vision from data","venue":null,"work_id":"1a0ed6c9-8943-4efc-a520-fc6830364198","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.265791Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:5ac8d9dbb877a6fb7b82fe30a49106f163fbd3d81f54852982084614f5395c45","observation_id":"3cd798c9-eaeb-4969-820e-39ad73b97f21","resolution":{"observed_at":"2026-08-10T20:08:46.670835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.646014Z","title":"Expert-level detection of pathologies from unannotated chest x-ray images via self- supervised learning","venue":null,"work_id":"a5b710a5-b9b0-430d-9797-bb0f73d238db","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.271490Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:a658cdd0794da24b425af24c0edcf4045799f899a33e6ff4f380879a3f52eaa4","observation_id":"bacd2688-c153-4551-9cd1-9d3ea0705ac0","resolution":{"observed_at":"2026-08-10T20:08:46.653013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.630506Z","title":"Language models are not naysayers: an anal- ysis of language models on negation benchmarks","venue":null,"work_id":"dbad2b9b-c631-4a2e-b45b-c1cfb3f88f20","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.277242Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f38fd640736a46a4ed0a475ccaf286605b0cd80d20f0ebee37d654306fe808c6","observation_id":"505de08f-ebd5-4b72-bb76-e06e2100f58f","resolution":{"observed_at":"2026-08-10T20:08:46.635529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.283190Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.283190Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:72db96c41222efa43278d7a903fcf268e0b483565b64ad018a8e5f6c450bf29b","observation_id":"57d0e0a7-a39f-419a-8995-ab824dc0e9ca","resolution":{"observed_at":"2026-08-10T20:08:46.283190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.601358Z","title":"Real-fake: Effective training data synthesis through distribution matching","venue":null,"work_id":"40b9cc11-3aec-4ae1-b01e-9cec90a1a20a","year":2024},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.289143Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:d346ff8acec62e5ad1240554a98177f2a9381a43fd1dfbef05697071326b9e0b","observation_id":"134bce19-b0fc-4a7e-8e3a-d53033da6fed","resolution":{"observed_at":"2026-08-10T20:08:46.607105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.582301Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? In ICLR, 2023","venue":null,"work_id":"db3346b7-98cf-4d93-8160-ba41baf00dd8","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.294239Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:13255b08e070368fbfc6fca3da1789b4e4835811cc41d5869b0d9bfb7651202a","observation_id":"28592488-7af3-4134-a2ea-ce92e7c9575c","resolution":{"observed_at":"2026-08-10T20:08:46.588431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.564560Z","title":"Lit: Zero-shot transfer with locked-image text tuning","venue":null,"work_id":"2919fd67-20c7-4656-aa37-29e7d393ccfc","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.298798Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:55659b9e9e7bf81516895c8b0993145b2f93f1575ff2dd13e99a62db72b027b1","observation_id":"56c7ca48-e364-4fab-afa8-8f56e3500ae9","resolution":{"observed_at":"2026-08-10T20:08:46.569731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.546392Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"1256601a-06c7-45ff-8811-065ed28acd2a","year":2023},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.304481Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:605f5f8d9568fe28a33370f37249389000767cd32a52004b7f3e9dbc5719bfd1","observation_id":"3b85b920-c49b-45cb-b1c9-3d6d2ee36ec1","resolution":{"observed_at":"2026-08-10T20:08:46.552358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.00915","last_updated":"2025-01-08T22:58:51Z","snapshot_observed_at":"2026-07-06T14:57:39.647497Z","submitted_at":"2023-03-02T02:20:04Z","title":"BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.00915","snapshot_observed_at":"2026-08-10T20:08:46.310828Z","title":"Biomedclip: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.310828Z"},"links":{"cited_paper":"/paper/2303.00915","citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:f37631ca1a198e0bf097e50f375d6bc0d76fb759f16fb5511176a60cd2e58444","observation_id":"5bdf1839-48ec-41c2-8e4d-f6c92c722db5","resolution":{"observed_at":"2026-08-10T20:08:46.310828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.511879Z","title":null,"venue":null,"work_id":"fe47149e-856c-4123-9e9b-0c3cdf0ee6da","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.321808Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:6010a72432dbe7a41e405f3f40fc93fe450cc8ef304df59ba8f9c64727050020","observation_id":"31792e35-f68e-4868-919a-52efb1ade9b3","resolution":{"observed_at":"2026-08-10T20:08:46.517398Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.493976Z","title":"Yes.” over “No","venue":null,"work_id":"852152c7-cfad-4c25-8def-d683040bdf05","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.326666Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:649f2bc362b3120f449ea55d1b113a67f4029b39ee89cec4f268627313c9af8f","observation_id":"eebb94a9-a692-433c-88eb-a37f3f06fbda","resolution":{"observed_at":"2026-08-10T20:08:46.500166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:47.491708Z","title":null,"venue":null,"work_id":"5999ddac-27e7-44df-9293-3ff311f5ef6c","year":2022},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.025541Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:161decbc47589250ed9695eb0591ed11e028c4d192cde1b8a22c9759e74e2aea","observation_id":"6ff600fa-2015-4ed7-996a-b3c7c3a277ae","resolution":{"observed_at":"2026-08-10T20:08:47.498368Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:08:46.529211Z","title":null,"venue":null,"work_id":"6972a785-f33f-4262-88c2-3dc62034e3e1","year":null},"citing_paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-10T20:08:46.316277Z"},"links":{"citing_paper":"/paper/2501.09425"},"observation_digest":"sha256:51511be0c3fa0fd269bab1f893a64c64325429cab6397bdf8191e26e8ca2674d","observation_id":"1367e721-a6e2-480a-80d9-672a22dd9629","resolution":{"observed_at":"2026-08-10T20:08:46.534065Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.09425","last_updated":"2025-05-13T06:30:11Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-11T03:14:03.476948Z","submitted_at":"2025-01-16T09:55:42Z","title":"Vision-Language Models Do Not Understand Negation"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":11,"verified_exact":0,"verified_fuzzy":44},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 8 inbound Pith citation observations for arXiv:2501.09425."}