{"as_of":"2026-08-21T08:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5a492b6e93152b39dde078d4d2e2b90076d1d7ea57f003b1311d2279967beb08","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":69,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":69,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":69,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:59:52.486750Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T19:36:29.281490Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-12T15:08:54.627473Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.15236","last_updated":"2024-11-21T23:37:03Z","snapshot_observed_at":"2026-08-17T16:26:41.425114Z","submitted_at":"2024-11-21T23:37:03Z","title":"Text Embedding is Not All You Need: Attention Control for Text-to-Image Semantic Alignment with Text Self-Attention Maps","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T15:08:54.627473Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2411.15236"},"observation_digest":"sha256:9c733f54112a57e23486be49fc011e4518f43df25575a032f47aaee43fab8cec","observation_id":"1c5deeee-8208-4051-8fa1-6835cb49f8d4","resolution":{"observed_at":"2026-08-12T15:08:54.627473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-11T22:24:06.780353Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.03513","last_updated":"2024-12-07T13:01:13Z","snapshot_observed_at":"2026-08-17T23:30:11.372290Z","submitted_at":"2024-12-04T17:56:49Z","title":"Enhancing CLIP Conceptual Embedding through Knowledge Distillation","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T22:24:06.780353Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2412.03513"},"observation_digest":"sha256:f60dd6d2d4d535a5e6e57f335cc29c8f8ba9beca19547c54b70f20d35cb6b951","observation_id":"96a26a43-d54f-4170-b4aa-97439c27ece7","resolution":{"observed_at":"2026-08-11T22:24:06.780353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-11T21:29:26.523972Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.04378","last_updated":"2025-05-08T19:16:29Z","snapshot_observed_at":"2026-08-14T18:05:39.883853Z","submitted_at":"2024-12-05T17:54:27Z","title":"VladVA: Discriminative Fine-tuning of LVLMs","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T21:29:26.523972Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2412.04378"},"observation_digest":"sha256:a0fe394798211f4902ee85d12f361bef1185e6ded8618920d480ec515517c715","observation_id":"50cdc964-c232-4af0-8d39-bb20326e98bb","resolution":{"observed_at":"2026-08-11T21:29:26.523972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-11T12:59:14.424665Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.13652","last_updated":"2025-03-25T09:06:08Z","snapshot_observed_at":"2026-08-12T21:38:59.632066Z","submitted_at":"2024-12-18T09:31:06Z","title":"RelationField: Relate Anything in Radiance Fields","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-11T12:59:14.424665Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2412.13652"},"observation_digest":"sha256:cfa4617d29e6654af2a318e2c4200293a96741009794119afc6ad96007f03098","observation_id":"877fa6fe-ab78-42a7-8366-1b8646e0b093","resolution":{"observed_at":"2026-08-11T12:59:14.424665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-10T22:32:38.758354Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.01366","last_updated":"2025-07-07T20:17:31Z","snapshot_observed_at":"2026-08-16T03:57:19.339413Z","submitted_at":"2025-01-02T17:20:41Z","title":"ViGiL3D: A Linguistically Diverse Dataset for 3D Visual Grounding","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-10T22:32:38.758354Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2501.01366"},"observation_digest":"sha256:53d4b7fc2fce4422d976d2acea78721d86453f46cd3f01df3b9c74dd16f584bf","observation_id":"1c27b363-2f48-49de-8803-0715862d5e01","resolution":{"observed_at":"2026-08-10T22:32:38.758354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-10T21:26:00.653529Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.04931","last_updated":"2025-06-27T10:07:29Z","snapshot_observed_at":"2026-08-19T17:06:16.165573Z","submitted_at":"2025-01-09T02:47:01Z","title":"Jailbreaking Multimodal Large Language Models via Shuffle Inconsistency","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T21:26:00.653529Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2501.04931"},"observation_digest":"sha256:3ac3cbc8bef1d8613051ab0f1435a74de25fc52345c02347874925694f6a4a2b","observation_id":"29cd9db1-eda2-4a78-ad7d-ccab212b63f4","resolution":{"observed_at":"2026-08-10T21:26:00.653529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-10T19:39:25.093138Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.09887","last_updated":"2025-01-17T00:18:34Z","snapshot_observed_at":"2026-08-16T21:00:30.842378Z","submitted_at":"2025-01-17T00:18:34Z","title":"FLORA: Formal Language Model Enables Robust Training-free Zero-shot Object Referring Analysis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T19:39:25.093138Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2501.09887"},"observation_digest":"sha256:cc80310ca6f06f6cf873080c2e10ab71cfb20f448ca83471f2cea4fc564d26bd","observation_id":"1a69c6cc-4c96-4ed1-b5c7-fb92fb010c16","resolution":{"observed_at":"2026-08-10T19:39:25.093138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2503.07265","last_updated":"2026-06-02T17:11:50Z","snapshot_observed_at":"2026-08-16T12:51:32.079370Z","submitted_at":"2025-03-10T12:47:53Z","title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-15T16:24:27.407376Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2503.07265"},"observation_digest":"sha256:689560a222b427022bd9b7a5898a513ce03aaf86ca4184b43a79b1c625b9858c","observation_id":"a402747d-6017-4b4e-a935-9af29a8907bf","resolution":{"observed_at":"2026-05-15T16:24:27.539998Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-16T10:59:52.486750Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.16801","last_updated":"2025-08-26T03:13:51Z","snapshot_observed_at":"2026-08-18T10:39:24.881540Z","submitted_at":"2025-04-23T15:20:53Z","title":"Decoupled Global-Local Alignment for Improving Compositional Understanding","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-16T10:59:52.486750Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2504.16801"},"observation_digest":"sha256:2755cf62f74300fab70009c61b3e601d389513be60c043d8221d62b335a5d002","observation_id":"4323f224-8df8-4547-8e42-746b7bad621f","resolution":{"observed_at":"2026-08-16T10:59:52.486750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-16T04:43:11.456920Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00759","last_updated":"2025-05-12T20:46:35Z","snapshot_observed_at":"2026-08-16T09:38:06.751547Z","submitted_at":"2025-05-01T17:47:55Z","title":"Multi-Modal Language Models as Text-to-Image Model Evaluators","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-16T04:43:11.456920Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2505.00759"},"observation_digest":"sha256:b988177c26f8703c5bdbdeecb2271faf876198c5f4e088e6e8f1ce639cb3d724","observation_id":"44242c63-bf64-44cd-9389-36287455d129","resolution":{"observed_at":"2026-08-16T04:43:11.456920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-15T21:55:28.262510Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.08622","last_updated":"2025-07-21T05:47:57Z","snapshot_observed_at":"2026-08-18T16:08:43.298691Z","submitted_at":"2025-05-13T14:40:22Z","title":"Visually Guided Decoding: Gradient-Free Hard Prompt Inversion with Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T21:55:28.262510Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2505.08622"},"observation_digest":"sha256:0b209613b2fc386d30666b35a49afe509011561a5e94a004eb01f3ce6716bf7f","observation_id":"d722abd3-f017-4e8d-a970-bc9b23a064c6","resolution":{"observed_at":"2026-08-15T21:55:28.262510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-07T15:17:42.791864Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15576","last_updated":"2025-08-28T04:15:35Z","snapshot_observed_at":"2026-08-18T16:08:16.947286Z","submitted_at":"2025-05-21T14:28:43Z","title":"Visual Perturbation and Adaptive Hard Negative Contrastive Learning for Compositional Reasoning in Vision-Language Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:17:42.791864Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2505.15576"},"observation_digest":"sha256:23513ae86e332a07882d9aad32b4159a8569be8a82082764b347f9063431b637","observation_id":"dff164eb-6c6a-4acf-8cb4-bee651993ea6","resolution":{"observed_at":"2026-08-07T15:17:42.791864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-07T14:35:00.457612Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18434","last_updated":"2025-05-24T00:02:48Z","snapshot_observed_at":"2026-08-13T07:05:13.469833Z","submitted_at":"2025-05-24T00:02:48Z","title":"TNG-CLIP:Training-Time Negation Data Generation for Negation Awareness of CLIP","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T14:35:00.457612Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2505.18434"},"observation_digest":"sha256:5c04253d9390eb70f202dcfaeb46b6f76b909907836702a620cc0a3e9d650718","observation_id":"3b905e8a-b3f0-4abc-a0ef-56ac6cc61eab","resolution":{"observed_at":"2026-08-07T14:35:00.457612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-07T13:21:12.731024Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it?arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.22079","last_updated":"2025-05-28T08:00:18Z","snapshot_observed_at":"2026-08-20T22:31:43.603798Z","submitted_at":"2025-05-28T08:00:18Z","title":"Bringing CLIP to the Clinic: Dynamic Soft Labels and Negation-Aware Learning for Medical Analysis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T13:21:12.731024Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2505.22079"},"observation_digest":"sha256:f3b09f41417e5c250af31f40746076f36f862994cc0cc87d91c824cbb2ae8824","observation_id":"5978c2a7-5ae2-4d2b-a6dc-1796a0f67198","resolution":{"observed_at":"2026-08-07T13:21:12.731024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-07T13:15:56.159838Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.22305","last_updated":"2025-05-28T12:41:08Z","snapshot_observed_at":"2026-08-10T16:45:41.991027Z","submitted_at":"2025-05-28T12:41:08Z","title":"IKIWISI: An Interactive Visual Pattern Generator for Evaluating the Reliability of Vision-Language Models Without Ground Truth","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T13:15:56.159838Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2505.22305"},"observation_digest":"sha256:a40f246e457725d15169da9859222a1eef7ff230ee1f5a13cf951c808909414a","observation_id":"2650a0af-46a6-4c1d-a009-fa9c1ab10f53","resolution":{"observed_at":"2026-08-07T13:15:56.159838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-07T12:03:53.099753Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.00718","last_updated":"2025-05-31T21:35:54Z","snapshot_observed_at":"2026-08-21T01:52:39.816168Z","submitted_at":"2025-05-31T21:35:54Z","title":"From Local Cues to Global Percepts: Emergent Gestalt Organization in Self-Supervised Vision Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:53.099753Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2506.00718"},"observation_digest":"sha256:3fd972c082018de1b1cec081930f1a97b8677f9d018cd53fb14c09c976043874","observation_id":"21d5b03d-47ce-423f-916d-6834d1354101","resolution":{"observed_at":"2026-08-07T12:03:53.099753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-07T10:28:36.339325Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05146","last_updated":"2025-06-19T07:19:46Z","snapshot_observed_at":"2026-08-18T16:08:13.359360Z","submitted_at":"2025-06-05T15:27:16Z","title":"CIVET: Systematic Evaluation of Understanding in VLMs","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T10:28:36.339325Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2506.05146"},"observation_digest":"sha256:86efe5d01a9ed05b34f5816dd438ac26eb09aa7e30ad96c9089994b56554a28c","observation_id":"ceca9211-8036-4760-80a7-1391c7ff2020","resolution":{"observed_at":"2026-08-07T10:28:36.339325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-06T20:11:51.695389Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.03683","last_updated":"2025-07-04T16:03:31Z","snapshot_observed_at":"2026-08-14T09:42:45.374800Z","submitted_at":"2025-07-04T16:03:31Z","title":"On the rankability of visual embeddings","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T20:11:51.695389Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2507.03683"},"observation_digest":"sha256:5e452e7ea9d70d4d2e6d5b5de1a1419c8faca13c552f3a28b864ee2389ff45f8","observation_id":"3ed8fddf-ba6d-48cb-8c12-f3952ea698e1","resolution":{"observed_at":"2026-08-06T20:11:51.695389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-06T18:48:45.958345Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07317","last_updated":"2025-07-28T17:21:56Z","snapshot_observed_at":"2026-08-13T09:39:02.931949Z","submitted_at":"2025-07-09T22:29:47Z","title":"ADIEE: Automatic Dataset Creation and Scorer for Instruction-Guided Image Editing Evaluation","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T18:48:45.958345Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2507.07317"},"observation_digest":"sha256:e7f05c59591c9fdc9ac81720c363f4a3e2a5c561abaa20d88274ca7c9f7691fe","observation_id":"48d636dd-8c85-446d-acd7-e7f440c730b8","resolution":{"observed_at":"2026-08-06T18:48:45.958345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-06T18:32:48.147656Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.08000","last_updated":"2025-07-10T17:59:59Z","snapshot_observed_at":"2026-08-13T09:31:31.845625Z","submitted_at":"2025-07-10T17:59:59Z","title":"Impact of Pretraining Word Co-occurrence on Compositional Generalization in Multimodal Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T18:32:48.147656Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2507.08000"},"observation_digest":"sha256:a1d67b34c8a5ac5ea4a72f46fe9f4d85dbd9d1a2c1fead06793eaf965cfb6613","observation_id":"e72a3b76-2f3c-414e-b73d-298ca1be321a","resolution":{"observed_at":"2026-08-06T18:32:48.147656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-06T18:24:54.449939Z","title":"When and why vision- language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.08513","last_updated":"2025-07-23T22:34:55Z","snapshot_observed_at":"2026-08-17T04:35:04.850485Z","submitted_at":"2025-07-11T12:00:10Z","title":"Advancing Multimodal LLMs by Large-Scale 3D Visual Instruction Dataset Generation","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T18:24:54.449939Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2507.08513"},"observation_digest":"sha256:c75a001b41032f09a0dcc8a377b31038956aa01cc615e999cf74c328fe26f077","observation_id":"720d82eb-b1da-4161-9cf6-1fba16e76b26","resolution":{"observed_at":"2026-08-06T18:24:54.449939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-06T18:36:01.402381Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.10442","last_updated":"2025-07-10T15:26:41Z","snapshot_observed_at":"2026-08-14T17:00:14.442606Z","submitted_at":"2025-07-10T15:26:41Z","title":"Response Wide Shut? Surprising Observations in Basic Vision Language Model Capabilities","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T18:36:01.402381Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2507.10442"},"observation_digest":"sha256:7a83a5de57ff71dddced02748f771fa665e6e67f36480100c2284ad18bd24d35","observation_id":"d715b9ac-d1c6-4571-9849-5ace6a338478","resolution":{"observed_at":"2026-08-06T18:36:01.402381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-15T18:08:58.775271Z","title":"When and why vision-language models be- have like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19064","last_updated":"2025-08-05T03:36:50Z","snapshot_observed_at":"2026-08-18T16:07:06.399663Z","submitted_at":"2025-07-25T08:25:48Z","title":"Negation-Aware Test-Time Adaptation for Vision-Language Models","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T18:08:58.775271Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2507.19064"},"observation_digest":"sha256:fc4fc93891eec82d9ef51e5a742de665ecb04ca41947b579057de97615406b1b","observation_id":"444a16d3-f9e4-41d5-9686-898887bac9e4","resolution":{"observed_at":"2026-08-15T18:08:58.775271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-06T12:08:01.848706Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it? arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.22100","last_updated":"2025-07-29T17:59:16Z","snapshot_observed_at":"2026-08-14T05:11:21.393015Z","submitted_at":"2025-07-29T17:59:16Z","title":"Trade-offs in Image Generation: How Do Different Dimensions Interact?","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T12:08:01.848706Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2507.22100"},"observation_digest":"sha256:c2506ab4cccd3c0d735cd53b6ab75e2980b009d7cad3f5797bba9d5fd0e54070","observation_id":"0e3e5cde-ade4-4b49-9635-629d88a904b2","resolution":{"observed_at":"2026-08-06T12:08:01.848706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-05T23:40:01.070516Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.05021","last_updated":"2025-08-07T04:15:08Z","snapshot_observed_at":"2026-08-18T16:08:17.864457Z","submitted_at":"2025-08-07T04:15:08Z","title":"MAG-Nav: Language-Driven Object Navigation Leveraging Memory-Reserved Active Grounding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T23:40:01.070516Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2508.05021"},"observation_digest":"sha256:b6445984295fa6c70da81b55e06f49c5e4fa7e631e9e0c2cb1f54b148e6ead51","observation_id":"5b0e5a8a-4ba5-4085-a5dc-479711bd9875","resolution":{"observed_at":"2026-08-05T23:40:01.070516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-05T17:45:39.596159Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.16687","last_updated":"2026-05-29T17:31:08Z","snapshot_observed_at":"2026-08-19T08:06:13.958785Z","submitted_at":"2025-08-21T18:29:17Z","title":"Native Hierarchical and Compositional Representations with Subspace Embeddings","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-05T17:45:39.596159Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2508.16687"},"observation_digest":"sha256:3d26363eaa4ce9d02364390326348f20b62b70e03ed8af3f4c1fb166ad9a3bf3","observation_id":"b0787d3e-1fb8-4ba3-8582-6cbabf57d47e","resolution":{"observed_at":"2026-08-05T17:45:39.596159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-05T11:56:10.958883Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.02175","last_updated":"2025-09-04T16:38:44Z","snapshot_observed_at":"2026-08-20T07:39:31.972350Z","submitted_at":"2025-09-02T10:32:58Z","title":"Understanding Space Is Rocket Science -- Only Top Reasoning Models Can Solve Spatial Understanding Tasks","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-05T11:56:10.958883Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2509.02175"},"observation_digest":"sha256:d181b5bf5b9a07eb562b60312d3d947763230297a9204afb146678c1550d2f93","observation_id":"a406c379-ad15-4816-8977-5b9deee8d4c0","resolution":{"observed_at":"2026-08-05T11:56:10.958883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-05T10:38:58.449457Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.03897","last_updated":"2025-09-12T05:44:35Z","snapshot_observed_at":"2026-08-18T15:24:17.009673Z","submitted_at":"2025-09-04T05:43:50Z","title":"SPECS: Specificity-Enhanced CLIP-Score for Long Image Caption Evaluation","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T10:38:58.449457Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2509.03897"},"observation_digest":"sha256:43a8f8faa79a5139b5d0de5fa52f1a0ee4086590b14c92d400ef6778b9c0ece3","observation_id":"fca59d4c-535f-4bc2-94db-56f283c6cb16","resolution":{"observed_at":"2026-08-05T10:38:58.449457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2509.19207","last_updated":"2026-05-12T09:10:11Z","snapshot_observed_at":"2026-08-11T13:03:49.442261Z","submitted_at":"2025-09-23T16:28:51Z","title":"Long Story Short: Disentangling Compositionality and Long-Caption Understanding in Contrastive VLMs","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-18T14:41:20.403259Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2509.19207"},"observation_digest":"sha256:473bc4d10c2af0edd043271bbeffe63504db333c15bff45a57101635cbf8dc00","observation_id":"f8da70f2-04a0-4eb4-ad7a-0cb71a05e33c","resolution":{"observed_at":"2026-05-18T14:41:30.190913Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2509.23370","last_updated":"2026-05-08T12:17:41Z","snapshot_observed_at":"2026-08-18T03:44:43.361421Z","submitted_at":"2025-09-27T15:36:59Z","title":"GRAPE: Let GRPO Supervise Query Rewriting by Ranking for Retrieval","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T12:18:06.259724Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2509.23370"},"observation_digest":"sha256:952d46d1754ee90600d2ce2d10646f3092f409b933acd64213852081860693d8","observation_id":"50be0de3-0ac9-46a6-a88a-d9493641a75e","resolution":{"observed_at":"2026-05-18T12:21:21.345749Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-04T13:52:17.224410Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?arXiv preprint arXiv:2210.01936,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.24566","last_updated":"2026-07-16T09:55:58Z","snapshot_observed_at":"2026-08-20T14:30:18.663894Z","submitted_at":"2025-09-29T10:19:22Z","title":"TokenSwap: Backdoor Attack on the Compositional Understanding of Large Vision-Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T13:52:17.224410Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2509.24566"},"observation_digest":"sha256:ba67ef87fb8e074f6d5579ba84da817dad5dbb641d22d0b37650b54e2c8d20f6","observation_id":"a026177b-d17f-4e53-a456-1f2c21502a7a","resolution":{"observed_at":"2026-08-04T13:52:17.224410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-03T21:09:38.983791Z","title":"Kecheng Zheng, Yifei Zhang, Wei Wu, Fan Lu, Shuailei Ma, Xin Jin, Wei Chen, and Yujun Shen","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.16527","last_updated":"2026-06-29T08:31:48Z","snapshot_observed_at":"2026-08-15T03:55:09.922667Z","submitted_at":"2025-11-20T16:41:36Z","title":"Contrastive vision-language learning with paraphrasing and negation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T21:09:38.983791Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2511.16527"},"observation_digest":"sha256:8a79d6a2a35f26264bc4259d6d1b3c71fa0494de93cbe27c2cb471c52895e8c8","observation_id":"bac79f28-4979-4612-8ae4-10343a704e77","resolution":{"observed_at":"2026-08-03T21:09:38.983791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2511.20814","last_updated":"2026-04-05T23:59:08Z","snapshot_observed_at":"2026-08-20T00:40:42.927681Z","submitted_at":"2025-11-25T20:00:47Z","title":"SPHINX: A Synthetic Environment for Visual Perception and Reasoning","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-17T04:19:26.808804Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2511.20814"},"observation_digest":"sha256:1dcb6e4e73e81122327124b904258ce12b721d00bd025c6994c4599f0e745fec","observation_id":"5df9f551-6620-434d-a92a-ae991542f506","resolution":{"observed_at":"2026-05-17T04:21:30.812869Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-03T19:22:51.598480Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2512.00918","last_updated":"2026-06-20T15:23:57Z","snapshot_observed_at":"2026-08-17T13:38:17.039028Z","submitted_at":"2025-11-30T14:52:11Z","title":"Sparse Neuron Ablation Triggers Catastrophic Collapse of the Language Core in Large Vision-Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T19:22:51.598480Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2512.00918"},"observation_digest":"sha256:3ef8b2bd3cf3446922c5f93235c74d9aa53b91f86aeb4cb655a2d3f7243ad9dd","observation_id":"60f5526d-83ee-4184-8ab0-574212d155b7","resolution":{"observed_at":"2026-08-03T19:22:51.598480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2512.13511","last_updated":"2026-05-01T11:23:38Z","snapshot_observed_at":"2026-08-13T14:28:42.446815Z","submitted_at":"2025-12-15T16:38:59Z","title":"Adapting MLLMs for Nuanced Video Retrieval","version":3},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-16T22:20:09.051957Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2512.13511"},"observation_digest":"sha256:c5f2242994429dae9aed421569207c5b2ccdb4993343df78adaa0177030ec02a","observation_id":"22650ba6-8ce3-4633-843e-02d983482f9a","resolution":{"observed_at":"2026-05-16T22:21:18.885153Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-03T05:29:03.873100Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.02220","last_updated":"2026-05-29T14:23:41Z","snapshot_observed_at":"2026-08-14T05:03:15.502680Z","submitted_at":"2026-02-02T15:26:19Z","title":"LangMap: A Human-Verified Benchmark for Hierarchical Open-Vocabulary Goal Navigation","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-03T05:29:03.873100Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2602.02220"},"observation_digest":"sha256:d5f035e0ee141545969e9ec63d61bef47cff09bac681f4581598b6928aa57ac7","observation_id":"fd44e1be-d98f-4032-8e77-aff850b1e7f6","resolution":{"observed_at":"2026-08-03T05:29:03.873100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-03T04:20:58.781116Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?arXiv preprint arXiv:2210.01936, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.05275","last_updated":"2026-07-01T09:38:37Z","snapshot_observed_at":"2026-08-15T20:10:49.368970Z","submitted_at":"2026-02-05T04:01:01Z","title":"Magic-MM-Embedding: Towards Visual-Token-Efficient Universal Multimodal Embedding with MLLMs","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-03T04:20:58.781116Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2602.05275"},"observation_digest":"sha256:ad0e6d0c790e65c3640e87aaed510fe60cd4ad4a350f2c9bebaad0ff7c7503da","observation_id":"ba776f33-078c-43d5-a1df-22b946cefd29","resolution":{"observed_at":"2026-08-03T04:20:58.781116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-15T13:27:51.848177Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?arXiv preprint arXiv:2210.01936,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.07109","last_updated":"2026-05-30T07:36:55Z","snapshot_observed_at":"2026-08-15T13:10:05.470489Z","submitted_at":"2026-03-07T08:40:09Z","title":"Vision Language Models Cannot Reason About Physical Transformation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-15T13:27:51.848177Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2603.07109"},"observation_digest":"sha256:9c45c611849bacdee57b3d296292b02b8aeb7450e4b810856191082f950d1901","observation_id":"3f5f8a49-aa49-4479-a17f-bc0fdcfb7987","resolution":{"observed_at":"2026-07-15T13:27:51.848177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2603.18373","last_updated":"2026-06-01T01:48:27Z","snapshot_observed_at":"2026-08-16T02:49:56.939545Z","submitted_at":"2026-03-19T00:15:05Z","title":"To See or To Please: Uncovering Visual Sycophancy and Split Beliefs in VLMs","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T09:18:34.853946Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2603.18373"},"observation_digest":"sha256:58b51058c0cfb6d30ee267efdcf740ea7537c41142e3b1549abf9f478cec51d4","observation_id":"24ebf0a7-2571-49e4-beb6-ea2d10b44608","resolution":{"observed_at":"2026-05-15T09:19:53.771783Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2604.03114","last_updated":"2026-04-03T15:36:00Z","snapshot_observed_at":"2026-08-14T17:20:26.822950Z","submitted_at":"2026-04-03T15:36:00Z","title":"Can VLMs Truly Forget? Benchmarking Training-Free Visual Concept Unlearning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T19:48:03.278822Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2604.03114"},"observation_digest":"sha256:863e8bc8fbc5555f9b80a7554d451dab72e52736db3351ad37c9aec04479b151","observation_id":"a984be64-eca3-412b-9df6-7fb35f0d700b","resolution":{"observed_at":"2026-05-13T19:48:11.172994Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2604.10604","last_updated":"2026-04-12T12:19:10Z","snapshot_observed_at":"2026-08-10T23:13:57.981652Z","submitted_at":"2026-04-12T12:19:10Z","title":"NSFL: A Post-Training Neuro-Symbolic Fuzzy Logic Framework for Boolean Operators in Neural Embeddings","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T15:52:36.777259Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2604.10604"},"observation_digest":"sha256:26ddeafa5814fdcb9f57d67a77f2242502a96712ea887a81e43b68d2fd244fc9","observation_id":"11950544-6bb9-47fc-b7f1-1b33fdc3dfb5","resolution":{"observed_at":"2026-05-11T09:41:04.121070Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2604.12335","last_updated":"2026-04-14T06:17:35Z","snapshot_observed_at":"2026-08-11T09:39:12.749021Z","submitted_at":"2026-04-14T06:17:35Z","title":"All in One: A Unified Synthetic Data Pipeline for Multimodal Video Understanding","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-05-10T15:26:55.369840Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2604.12335"},"observation_digest":"sha256:cb9616ee66c70685bb734011b82cc63e974ee1f40bae76d228ffc8f79e7e7186","observation_id":"a31ff7df-3579-42df-8726-b996d065e135","resolution":{"observed_at":"2026-05-11T10:31:03.898475Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2604.13313","last_updated":"2026-04-14T21:28:46Z","snapshot_observed_at":"2026-08-13T15:13:50.391868Z","submitted_at":"2026-04-14T21:28:46Z","title":"Concrete Jungle: Towards Concreteness Paved Contrastive Negative Mining for Compositional Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T15:24:57.169737Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2604.13313"},"observation_digest":"sha256:545bb94da2488af3b6e7ab963e305c76f973b7e3a4b4868a51fc58154b854d80","observation_id":"8ab51118-45e0-44e5-985c-f3a8e5f74b6e","resolution":{"observed_at":"2026-05-11T10:36:04.871204Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2604.20135","last_updated":"2026-04-22T03:02:21Z","snapshot_observed_at":"2026-08-15T04:42:31.218737Z","submitted_at":"2026-04-22T03:02:21Z","title":"AFMRL: Attribute-Enhanced Fine-Grained Multi-Modal Representation Learning in E-commerce","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-10T00:55:56.146885Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2604.20135"},"observation_digest":"sha256:68732506e63d2a574c1089292ec2bf5800e350c2485b886d53c69d3bfe6d80e9","observation_id":"cc21d4ef-fd35-4d23-8679-164dbccdc380","resolution":{"observed_at":"2026-05-10T00:59:49.741519Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.06512","last_updated":"2026-05-07T16:22:21Z","snapshot_observed_at":"2026-08-15T16:32:41.066580Z","submitted_at":"2026-05-07T16:22:21Z","title":"DCR: Counterfactual Attractor Guidance for Rare Compositional Generation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-08T13:06:56.964670Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.06512"},"observation_digest":"sha256:9d6569ea86c5311ea470a7a920d80072acaf1482e9151b03e91146269e34950d","observation_id":"5fd61976-d684-4fb7-a2b1-815f2c3b182c","resolution":{"observed_at":"2026-05-11T19:01:11.745909Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.11107","last_updated":"2026-05-11T18:13:05Z","snapshot_observed_at":"2026-08-11T03:45:15.335189Z","submitted_at":"2026-05-11T18:13:05Z","title":"Birds of a Feather Flock Together: Background-Invariant Representations via Linear Structure in VLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-13T07:26:02.947081Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.11107"},"observation_digest":"sha256:0f51f67d49be7da8397ab9af517d2f4655b7703dc0dec1c48f8af6885f13427c","observation_id":"8ac142af-b99a-4ded-bb73-7c7862bdc22d","resolution":{"observed_at":"2026-05-13T07:27:29.700340Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.11558","last_updated":"2026-05-12T05:41:36Z","snapshot_observed_at":"2026-08-17T06:15:00.618167Z","submitted_at":"2026-05-12T05:41:36Z","title":"A Composite Activation Function for Learning Stable Binary Representations","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-13T02:03:42.456988Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.11558"},"observation_digest":"sha256:58b3fde38b38ade31444e031526879ebf3eebd883391172c6be0b659a9e7262c","observation_id":"29fa1dec-d6fb-4b8e-a0f3-2a0850ee0988","resolution":{"observed_at":"2026-05-13T02:07:07.934212Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.12485","last_updated":"2026-05-18T04:19:54Z","snapshot_observed_at":"2026-08-17T09:14:38.511957Z","submitted_at":"2026-05-12T17:58:22Z","title":"Letting the neural code speak: Automated characterization of monkey visual neurons through human language","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-05-13T02:12:00.335761Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.12485"},"observation_digest":"sha256:817258ef1339cd3a38e3a62d649af970dd68c52363a184ef704c817284c24592","observation_id":"e9375fac-c45b-4910-824a-7ea38570c089","resolution":{"observed_at":"2026-05-13T02:12:07.069071Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.12485","last_updated":"2026-05-18T04:19:54Z","snapshot_observed_at":"2026-08-17T09:14:38.511957Z","submitted_at":"2026-05-12T17:58:22Z","title":"Letting the neural code speak: Automated characterization of monkey visual neurons through human language","version":2},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-05-20T21:17:53.615009Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.12485"},"observation_digest":"sha256:8f98438699604b6291691da1d1b88ceb4cf820d2cf99759e277e790a15a4dfd5","observation_id":"deb95c45-a754-48f7-aa21-abff4ff4c1fb","resolution":{"observed_at":"2026-05-20T21:19:02.977300Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.12872","last_updated":"2026-05-13T01:36:43Z","snapshot_observed_at":"2026-08-14T11:08:19.952564Z","submitted_at":"2026-05-13T01:36:43Z","title":"SMA: Submodular Modality Aligner For Data Efficient Multimodal Learning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-14T20:32:00.361991Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.12872"},"observation_digest":"sha256:9186bc269806362779be2556ef32ba5e8b4f4858e143a339c29d34c8d4c0592b","observation_id":"74997e67-f9ae-4f3a-9c1a-35e44398aed0","resolution":{"observed_at":"2026-05-14T20:32:56.639657Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.13047","last_updated":"2026-05-13T06:11:47Z","snapshot_observed_at":"2026-08-14T22:10:00.883049Z","submitted_at":"2026-05-13T06:11:47Z","title":"Revealing the Gap in Human and VLM Scene Perception through Counterfactual Semantic Saliency","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-14T20:35:08.286755Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.13047"},"observation_digest":"sha256:94156ce279b24cce98c96991d30005a7caea89414134a882b37403a82ec923ae","observation_id":"73d649b2-e73c-4c8d-961c-d21814af32b2","resolution":{"observed_at":"2026-05-14T20:39:27.778099Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.16403","last_updated":"2026-05-13T05:00:19Z","snapshot_observed_at":"2026-07-06T23:27:34.514541Z","submitted_at":"2026-05-13T05:00:19Z","title":"When Vision Speaks for Sound","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-20T22:12:52.160596Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.16403"},"observation_digest":"sha256:9e4188569de0c33364783a815492de5a41b85d34bb2725cc60801893873f2ab1","observation_id":"69e4e4ef-8b55-4479-bf43-c039682694d7","resolution":{"observed_at":"2026-05-20T22:13:46.772599Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.18018","last_updated":"2026-05-18T08:09:37Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T08:09:37Z","title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-20T12:10:54.874012Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.18018"},"observation_digest":"sha256:e051ef5abf0e7c668476a2c9d2d06970af4d99ea23ff37d9e4c7e21f11cb8802","observation_id":"3beef4a7-9412-4812-8e8b-6b6c89f5b183","resolution":{"observed_at":"2026-05-20T12:13:16.266930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2605.26396","last_updated":"2026-05-29T07:01:08Z","snapshot_observed_at":"2026-08-12T20:35:56.252174Z","submitted_at":"2026-05-25T23:59:02Z","title":"Advancing Creative Physical Intelligence in Large Multimodal Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T21:09:37.228058Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2605.26396"},"observation_digest":"sha256:444516c177ccce87e8f8c29185e71c960e2e94d806a2a70200221c2d5a78a98c","observation_id":"8d80a690-8f81-4c46-b93c-095273bc7ce0","resolution":{"observed_at":"2026-06-29T22:34:02.910384Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2606.07568","last_updated":"2026-05-26T02:19:47Z","snapshot_observed_at":"2026-08-13T16:10:49.659022Z","submitted_at":"2026-05-26T02:19:47Z","title":"A Systematic Study of Behavioral Cloning for Scientific Data Annotation","version":1},"reference_index":238,"source":"arxiv_source","source_observed_at":"2026-06-29T16:23:08.402194Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2606.07568"},"observation_digest":"sha256:8ce4ac94ba6a4796270a2c263fa7f256e99f1cac506dc57989e781f02f2a8adf","observation_id":"97ce575d-8ba8-4720-8fef-d05a526d1cd5","resolution":{"observed_at":"2026-06-29T16:23:38.982894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2606.19941","last_updated":"2026-06-18T08:39:07Z","snapshot_observed_at":"2026-08-03T00:29:25.767611Z","submitted_at":"2026-06-18T08:39:07Z","title":"Compositionality Emerges in a Narrow Depth-Connectivity Regime: Architecture Constraints and Solution Manifolds","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-26T18:22:56.676469Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2606.19941"},"observation_digest":"sha256:9ad66c14d293abcc06d4a3a00d9ea44c8e8739214aaf71a8acadaf1f19bd8e29","observation_id":"068f2855-f599-4563-8fce-1f71c2db2db8","resolution":{"observed_at":"2026-07-04T03:09:29.816466Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2606.20177","last_updated":"2026-06-22T09:23:26Z","snapshot_observed_at":"2026-08-14T06:54:59.459635Z","submitted_at":"2026-06-18T12:46:56Z","title":"Evaluating and Enhancing Negation Comprehension in Remote Sensing MLLMs","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T18:08:37.978569Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2606.20177"},"observation_digest":"sha256:aecf9f9e8539ff399081cf13564f65ef5fc5dd9987bbfe4017bc1d32cbca3262","observation_id":"0ec36d21-fe43-4b22-b26f-4a82aba2d2a5","resolution":{"observed_at":"2026-07-04T03:29:29.295857Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2606.26794","last_updated":"2026-06-25T09:27:54Z","snapshot_observed_at":"2026-08-16T14:44:55.956830Z","submitted_at":"2026-06-25T09:27:54Z","title":"ReasonCLIP-58M: Visually Grounded Commonsense Reasoning Supervision for CLIP","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-06-26T05:03:15.044146Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2606.26794"},"observation_digest":"sha256:23d14a8e2de30a706f4d4e7203803b9bdd682eec976796c92b48e5a499fefd6c","observation_id":"099a458a-b108-4c3e-a69f-0695d16a60e3","resolution":{"observed_at":"2026-07-04T13:39:50.738424Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2606.30638","last_updated":"2026-06-29T17:58:41Z","snapshot_observed_at":"2026-08-13T19:34:16.780226Z","submitted_at":"2026-06-29T17:58:41Z","title":"Open-Vocabulary and Referring Segmentation for 3D Gaussians Using 2D Detectors","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-30T05:53:26.394245Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2606.30638"},"observation_digest":"sha256:5d4019108c762512b7edb34145c5e8bfa25980674cd5a001f60086c22adb695e","observation_id":"855a6172-026e-477f-b46d-b12025a6dd25","resolution":{"observed_at":"2026-06-30T05:54:18.335565Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":"2210.01936","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-09T19:36:29.281490Z","title":"International Conference on Learning Representations (ICLR) , year =","venue":"cs.CV","work_id":"f61f18f6-7b0d-4a42-91d7-f7dc1efcd5bf","year":2022},"citing_paper":{"arxiv_id":"2607.07135","last_updated":"2026-07-13T05:58:12Z","snapshot_observed_at":"2026-08-20T14:40:04.174245Z","submitted_at":"2026-07-08T08:30:48Z","title":"Sparse Attention for Dense Open-Vocabulary Prediction in CLIP","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-09T19:26:49.805499Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2607.07135"},"observation_digest":"sha256:8816ea6f4f7b89e91fc0d496e3f8c385e33a4323fd5dc5e28ecec69ceff79b26","observation_id":"3ee4f559-ea08-47b3-9dc0-b4a6325d463b","resolution":{"observed_at":"2026-07-09T19:36:29.283063Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-14T15:51:57.429693Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.07135","last_updated":"2026-07-13T05:58:12Z","snapshot_observed_at":"2026-08-20T14:40:04.174245Z","submitted_at":"2026-07-08T08:30:48Z","title":"Sparse Attention for Dense Open-Vocabulary Prediction in CLIP","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-14T15:51:57.429693Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2607.07135"},"observation_digest":"sha256:18f7d448b1167b945d34ac10fe34000db80b967461a6952c84196830a138f799","observation_id":"6a0b2fbd-b3b5-42cb-b2ee-62af29f5c5bd","resolution":{"observed_at":"2026-07-14T15:51:57.429693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-02T05:13:02.469296Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13437","last_updated":"2026-07-15T04:40:36Z","snapshot_observed_at":"2026-08-17T03:28:19.903660Z","submitted_at":"2026-07-15T04:40:36Z","title":"CLIP-Guided Label-Free Discriminative Region Scoring for Fine-Grained Classification","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T05:13:02.469296Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2607.13437"},"observation_digest":"sha256:fe7a925853f1ff2ab07760cc1614512344790c6d66913a3f96a921a1058c51f6","observation_id":"9105c657-07af-48d0-9cae-0870952e771e","resolution":{"observed_at":"2026-08-02T05:13:02.469296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-01T23:14:57.326327Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15491","last_updated":"2026-07-16T22:39:19Z","snapshot_observed_at":"2026-08-18T16:08:15.038310Z","submitted_at":"2026-07-16T22:39:19Z","title":"Trajectory-aware Cross-view Geo-localization with Sequential Observations","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T23:14:57.326327Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2607.15491"},"observation_digest":"sha256:e55234566176f7d6fcf4414a5e2a2be3ff368935a47e531ee473dcc25f02c936","observation_id":"4164caff-3f99-47c4-8d2d-6a1baa7a8a81","resolution":{"observed_at":"2026-08-01T23:14:57.326327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-07-30T23:29:03.410996Z","title":"arXiv preprint arXiv:2210.01936 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26706","last_updated":"2026-07-29T09:52:20Z","snapshot_observed_at":"2026-08-16T20:11:36.756442Z","submitted_at":"2026-07-29T09:52:20Z","title":"TPD: Temporal Prior Decoupling for Text-to-Video Diffusion Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-07-30T23:29:03.410996Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2607.26706"},"observation_digest":"sha256:ad48f87672aea6f3a361ba0020975761b20277524f1d32917ea6c0db21ac6be9","observation_id":"27680bac-a6b1-4bf0-8faf-6e63a090ef6f","resolution":{"observed_at":"2026-07-30T23:29:03.410996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-15T15:24:47.569466Z","title":"arXiv preprint arXiv:2210.01936 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00726","last_updated":"2026-08-01T15:47:52Z","snapshot_observed_at":"2026-08-17T22:49:17.630687Z","submitted_at":"2026-08-01T15:47:52Z","title":"Foveated Probes Recover Localized Binding Information in Vision Foundation Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-15T15:24:47.569466Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2608.00726"},"observation_digest":"sha256:dd625592b587376de158ff1889ad7777e0d5cf57ee90271386461bb8f42aed5d","observation_id":"40d68c93-ccc3-4738-8008-65f4bbac032b","resolution":{"observed_at":"2026-08-15T15:24:47.569466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-15T15:10:13.240029Z","title":"2023.When and why vision-language models behave like bags-of-words, and what to do about it?arXiv:2210.01936 https://arxiv.org/abs/2210.01936","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01635","last_updated":"2026-08-03T03:07:54Z","snapshot_observed_at":"2026-08-18T02:02:56.674864Z","submitted_at":"2026-08-03T03:07:54Z","title":"Mitigating Visual Degradation in MLLMs via Spatial-Spectral Visual Anchor Learning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T15:10:13.240029Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2608.01635"},"observation_digest":"sha256:37cc6047e5eaab583966f5b28842ddeb7a5fe6fb5972cade08b7948423633948","observation_id":"0a9b5cbe-0297-4743-80ae-cc5f2ef9f23f","resolution":{"observed_at":"2026-08-15T15:10:13.240029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-04T17:29:19.221100Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01977","last_updated":"2026-08-03T09:41:18Z","snapshot_observed_at":"2026-08-15T22:57:39.253917Z","submitted_at":"2026-08-03T09:41:18Z","title":"SVGEval: A Vision-Grounded Framework for Perceptual-Quality Benchmarking and Evaluation in Text-to-SVG Generation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T17:29:19.221100Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2608.01977"},"observation_digest":"sha256:3c73d3fe2f3bc4a0cbea81351d91927fbdda68934f9b7154d3f8d3bb3a3b1011","observation_id":"8d2b224c-6208-4a48-b279-761cbc5e347b","resolution":{"observed_at":"2026-08-04T17:29:19.221100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-07T00:14:42.565379Z","title":"arXiv preprint arXiv:2210.01936 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04034","last_updated":"2026-08-03T02:28:52Z","snapshot_observed_at":"2026-08-09T14:43:30.524152Z","submitted_at":"2026-08-03T02:28:52Z","title":"A Multimodal Automatic Redteaming Evaluation based on Atomic Jailbreak Strategy Decoupling and Combination","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:42.565379Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2608.04034"},"observation_digest":"sha256:afab6809d264a383c3b3037575e9d13bfa5c1b8b323c45993f5ccd084a4aecc9","observation_id":"a2e2df52-ce0a-4a6e-95b3-26c4c3e16a94","resolution":{"observed_at":"2026-08-07T00:14:42.565379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01936","snapshot_observed_at":"2026-08-15T14:23:02.412149Z","title":"InProceedings of the IEEE/CVF International Conference on Computer Vision (ICCV), 21970–21980","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10524","last_updated":"2026-08-11T06:00:27Z","snapshot_observed_at":"2026-08-20T05:26:03.163088Z","submitted_at":"2026-08-11T06:00:27Z","title":"Rethinking Text-Based Image Retrieval in Specific Domain","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T14:23:02.412149Z"},"links":{"cited_paper":"/paper/2210.01936","citing_paper":"/paper/2608.10524"},"observation_digest":"sha256:cb9a21b3cd58a7fd68c291b3c4ecc6fcaace38378cf5d1b6f0a9427acad94a39","observation_id":"f7e661eb-b70a-42f9-85e1-b675b0a4ffa6","resolution":{"observed_at":"2026-08-15T14:23:02.412149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2210.01936/citation-record","integrity":"/paper/2210.01936/integrity","json":"/paper/2210.01936/citation-record.json","paper":"/paper/2210.01936"},"outbound":[],"paper":{"arxiv_id":"2210.01936","last_updated":"2023-03-23T23:21:38Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T00:46:15.165076Z","submitted_at":"2022-10-04T22:13:25Z","title":"When and why vision-language models behave like bags-of-words, and what to do about it?"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 69 inbound Pith citation observations for arXiv:2210.01936."}