{"as_of":"2026-08-20T17:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8d072b9a1b47f0f2e3c370385bba19a02e707a0e71411df803f4f30c2b5f20ec","coverage":[{"denominator":80,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":80,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T23:45:53.258575Z","state":"measured"},{"denominator":80,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":80,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.06280/citation-record","integrity":"/paper/2505.06280/integrity","json":"/paper/2505.06280/citation-record.json","paper":"/paper/2505.06280"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.537076Z","title":"Single-stage semantic segmentation from image labels","venue":null,"work_id":"5fd8dc6a-0466-4944-b483-089cbfc11ddb","year":2020},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.913097Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:11559d6640dc3c372ee1b31d301ca25b4a11822d8f7e5d51435c6dee547bfb07","observation_id":"f07a75a6-1f04-44df-9e77-4799db1fe2b4","resolution":{"observed_at":"2026-08-15T23:45:54.542048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.521738Z","title":"Enhancing open-vocabulary semantic seg- mentation with prototype retrieval","venue":null,"work_id":"fcbba57f-24e7-49bc-afa1-06c13c2c9ebf","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.917865Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:da38e48a99b7367d21879489120cb189f50fd3d1c23cd51189f74d245d809ff5","observation_id":"0800f2fd-2408-4651-bb73-d2b737b94eeb","resolution":{"observed_at":"2026-08-15T23:45:54.526264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.507208Z","title":"Zerowaste dataset: To- wards deformable object segmentation in cluttered scenes","venue":null,"work_id":"4370c7a2-8d11-410b-a91c-8c43b59efbb8","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.922215Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:cc89273d7f8fbd45352a2261b777a0eb70fd65b24bafe14494c7df0269223ddb","observation_id":"493c422a-fecd-45b5-b9e4-7f12709acdbd","resolution":{"observed_at":"2026-08-15T23:45:54.511602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.492931Z","title":"What a mess: Multi-domain evaluation of zero-shot semantic segmentation.Advances in Neural Infor- mation Processing Systems, 36:73299–73311, 2023","venue":null,"work_id":"661177c8-e96b-4e95-b6b1-ea1ddc27eb71","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.926542Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:eb9e6c966f2eecffaa2b6ab8ee59e71b8ae51983ba6ad489f831f401aae42d20","observation_id":"55d548c3-5d63-4254-98e4-f9fe429a3555","resolution":{"observed_at":"2026-08-15T23:45:54.497495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.478438Z","title":"Zero-shot semantic segmentation.NeurIPS, 32, 2019","venue":null,"work_id":"b284b939-5c57-4852-a747-9bf36a06cb01","year":2019},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.931433Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:d67f4f40ca904986dd8fb19a9b792f17506f98ae0531bc94270c3a7913ea92a3","observation_id":"72edb67f-8192-4116-9b55-957b71213c55","resolution":{"observed_at":"2026-08-15T23:45:54.483241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.464116Z","title":"Emerg- ing properties in self-supervised vision transformers","venue":null,"work_id":"d8b7b773-46bb-447b-9112-ecbb2fdc1759","year":2021},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.935722Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:ae1c478b1ce3967303a15fefd55765ff543781b9785f5832922e8b8ba4d6e351","observation_id":"d100964b-1be7-44fb-95aa-ac3a87d74384","resolution":{"observed_at":"2026-08-15T23:45:54.468864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:52.940369Z","title":"Modeling the background for incremental learning in semantic segmentation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.940369Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:267864e49fa0838fb1b23c011778c9cc742aeaa9b4695c5b3ea9d7b978e38e17","observation_id":"3bc0245b-b6ea-43a1-9bb1-7984eba22b84","resolution":{"observed_at":"2026-08-15T23:45:52.940369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.01415","last_updated":"2021-10-18T16:51:31Z","snapshot_observed_at":"2026-08-17T16:37:37.133909Z","submitted_at":"2020-11-30T20:45:56Z","title":"Prototype-based Incremental Few-Shot Semantic Segmentation","version":2},"cited_work":{"arxiv_id":"2012.01415","doi":null,"metadata_source":"pith","pith_arxiv_id":"2012.01415","snapshot_observed_at":"2026-08-15T23:45:53.445473Z","title":"Prototype-based Incremental Few-Shot Semantic Segmentation","venue":"cs.CV","work_id":"949a8cbc-92d2-42ff-9442-104c04d45528","year":2020},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.944637Z"},"links":{"cited_paper":"/paper/2012.01415","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:e027d2a073e1a58bd238abcad50a0049dc4e1b4dd2d6341656fbe1791f96b666","observation_id":"0416b7bc-0cb0-430b-bd6d-b639ad0406ba","resolution":{"observed_at":"2026-08-15T23:45:53.452406Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.440221Z","title":"Learn- ing to generate text-grounded mask for open-world semantic segmentation from only image-text pairs","venue":null,"work_id":"d41a1672-039e-4ab6-8b74-b161e1ad05f6","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.949342Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:462c755c572ae8e611fbc923d1ef4a6ea015c4e567bd95b196009ef7168c23d0","observation_id":"fb2715f5-d1cd-4d8c-8b11-00790a14fe48","resolution":{"observed_at":"2026-08-15T23:45:54.445016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.14735","last_updated":"2025-05-11T09:23:41Z","snapshot_observed_at":"2026-08-18T20:37:49.047473Z","submitted_at":"2023-10-23T09:15:18Z","title":"Unleashing the potential of prompt engineering for large language models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.14735","snapshot_observed_at":"2026-08-15T23:45:52.953664Z","title":"Unleashing the potential of prompt engineer- ing in large language models: a comprehensive review.arXiv preprint arXiv:2310.14735, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.953664Z"},"links":{"cited_paper":"/paper/2310.14735","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:e245c763282181296f477870d99d69e335444fac092f5ef0ce377ace8e33f967","observation_id":"0800f009-6dd0-425c-89a8-ee02a5a21550","resolution":{"observed_at":"2026-08-15T23:45:52.953664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.425510Z","title":"Exploring open-vocabulary semantic segmentation from clip vision encoder distillation only","venue":null,"work_id":"3f9bbf1b-da4c-43b9-84fd-359ff77127a5","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.957730Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:93266e940ce15f4e44fe05563e8abb6d31dab0d71398104525a23f5627b644cb","observation_id":"629638ae-b1cb-4269-8908-33a766f4e94e","resolution":{"observed_at":"2026-08-15T23:45:54.430278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.410489Z","title":"MMSegmentation: Openmmlab semantic segmentation toolbox and benchmark.https : / / github","venue":null,"work_id":"8ed191ec-3b18-4ae4-ab27-c25194dcd68e","year":2020},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.961881Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:eb8cfc2ac7183a8353528364df9d4612c46dcaa22ecb97b2ee0fcc6ceb056e87","observation_id":"3d058f83-16ab-44d5-a5f8-5c99730f166f","resolution":{"observed_at":"2026-08-15T23:45:54.415153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:52.966278Z","title":"The cityscapes dataset for semantic urban scene understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.966278Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:3f4dd241400e0a712208772fbb1f5d5910e3a608f7a69e7dd126db2d4a9082d2","observation_id":"048fe4ce-5094-487e-8f18-171b37526b63","resolution":{"observed_at":"2026-08-15T23:45:52.966278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-16T09:25:53.087782Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-15T23:45:52.970477Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale.arXiv preprint arXiv:2010.11929, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.970477Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:52a2c0cd6e79e0928a0d2a449ef0f047c9b537d58f81ec6a336f9f78f3ee8699","observation_id":"f195cb40-260f-48d6-8ecd-4cce337cb13a","resolution":{"observed_at":"2026-08-15T23:45:52.970477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.385665Z","title":"A new large- scale food image segmentation dataset and its application to food calorie estimation based on grains of rice","venue":null,"work_id":"60c2b557-a32d-42be-a594-840501900e68","year":2019},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.975040Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:118b2a5ba88109bc6037ba4d5c3c7b27f4c4451b99296c2309833638611588bc","observation_id":"77534581-3c24-4219-9bf5-20f05bbd47c1","resolution":{"observed_at":"2026-08-15T23:45:54.390642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.371185Z","title":"The pascal visual object classes (voc) challenge.International journal of computer vision, 88:303–338, 2010","venue":null,"work_id":"aaa74226-23c2-40f7-aace-cc4325a7b865","year":2010},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.979253Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:ab447190e83da4bfa1829acb4b8b59e33d6f92b79589435b66c4562c1dcf5b6f","observation_id":"68302ed3-8a0a-4b2d-b7f4-73a71b3a88db","resolution":{"observed_at":"2026-08-15T23:45:54.375963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.356873Z","title":"Model- agnostic meta-learning for fast adaptation of deep networks","venue":null,"work_id":"0e0d2779-8cbe-4448-8275-bd1eff8aa389","year":2017},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.983364Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:80d58b67b0f19d348e919e2a849938da7d68453a7585d556b7ada6fcc244d10b","observation_id":"0bdab561-d681-45ce-a482-26108d0b1d83","resolution":{"observed_at":"2026-08-15T23:45:54.361612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.342348Z","title":"Scal- ing open-vocabulary image segmentation with image-level labels","venue":null,"work_id":"bbc031c8-2891-43ae-a38f-a1e47fed81d5","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.987558Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:7dafcb6dfff3e75884380cd26908f1cb51d78e8799e59e666ae55cd1537e04c1","observation_id":"7f149794-cd90-4c0b-a5d0-7565706c9d3a","resolution":{"observed_at":"2026-08-15T23:45:54.347185Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.327509Z","title":"Context-aware feature generation for zero- shot semantic segmentation","venue":null,"work_id":"85b986b3-ff50-4368-99c6-0cb23b9ff009","year":2020},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.991656Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:9242f4aa89e811c67c4dae1941c899856aee69394c30c6e11a19c04a973eba8b","observation_id":"806dd133-c184-4104-8a42-42a1a2043644","resolution":{"observed_at":"2026-08-15T23:45:54.332169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.313402Z","title":"Mvp-seg: Multi-view prompt learning for open-vocabulary semantic segmentation","venue":null,"work_id":"c6a27c60-635c-455c-a00a-ddbb23f6cc91","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:52.996000Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:01aed1198f6a09e70320947f2ed4e64a9fb08850208dfb9a6fb92f392e41d8c3","observation_id":"54afdd9b-f820-4eff-b87a-fc1d7939c084","resolution":{"observed_at":"2026-08-15T23:45:54.317850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08181","last_updated":"2024-09-16T20:11:48Z","snapshot_observed_at":"2026-08-16T14:01:33.981371Z","submitted_at":"2024-04-12T01:08:04Z","title":"Pay Attention to Your Neighbours: Training-Free Open-Vocabulary Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08181","snapshot_observed_at":"2026-08-15T23:45:53.000543Z","title":"Pay attention to your neighbours: Training-free open-vocabulary semantic segmentation.arXiv preprint arXiv:2404.08181, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.000543Z"},"links":{"cited_paper":"/paper/2404.08181","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:9a159a2089cf46ef0f81128c2ce89df013dbfff00d9d85cb1825815664cedab0","observation_id":"a20e0fc1-0908-4a8a-912a-194482eb7b9a","resolution":{"observed_at":"2026-08-15T23:45:53.000543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.298536Z","title":"Cost aggregation with 4d convolutional swin transformer for few-shot segmentation","venue":null,"work_id":"13d222b4-d4dd-4546-a25d-dcb2913bdeb5","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.005050Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:6f446446ec897de449b7cfe184c75edff5b1b81d6cc625018ef657dec8f1c29e","observation_id":"4a620cf5-650a-4cd1-b63f-24ac029e4e96","resolution":{"observed_at":"2026-08-15T23:45:54.303219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.283385Z","title":"Open- clip.https : / / github","venue":null,"work_id":"12b3e72d-906d-47fc-bc8d-235275573baa","year":2021},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.009515Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:a813354b7ed6751336d06ed1d8cc8f7465c8e591fd6e44a1b425f3afde530d24","observation_id":"97c5f418-c224-4789-89d1-73d66cc6cc39","resolution":{"observed_at":"2026-08-15T23:45:54.288480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.268482Z","title":"Diffusion models for zero-shot open-vocabulary segmentation.arXiv e-prints, pages arXiv–2306, 2023","venue":null,"work_id":"50522d86-32e4-4498-a7ad-255437891709","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.013648Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:abb94379a4b1168b94360b5802f420a32fbd95832dfa158d981dd40cf0487a30","observation_id":"182b3a74-4977-4807-a8ca-7d8249cdda60","resolution":{"observed_at":"2026-08-15T23:45:54.273110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.252616Z","title":"Segment any- thing","venue":null,"work_id":"1a276ba5-fd3b-45a4-8151-0e81f4cb7aee","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.017759Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:92d481dbfc43c5d9d6b5a4d5bf921589ce2b47285f893929b0ad3dcaa9f15cd6","observation_id":"6da7114c-8038-46a0-8132-62a479f290d0","resolution":{"observed_at":"2026-08-15T23:45:54.257532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.238065Z","title":"Overcoming catastrophic forgetting in neu- ral networks.Proceedings of the national academy of sci- ences, 114(13):3521–3526, 2017","venue":null,"work_id":"ad9ffc39-6074-452f-822e-9a14ee8018ef","year":2017},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.021841Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:0abd8768af97e392986acda59ec6b605dbd191a1d12b7ed46cef400c67abe89e","observation_id":"b5050f5b-42ac-4171-8050-fb0c1f3ce9b1","resolution":{"observed_at":"2026-08-15T23:45:54.242687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.223027Z","title":"Proxyclip: Proxy at- tention improves clip for open-vocabulary segmentation","venue":null,"work_id":"e166d2e1-3958-41ef-b5a0-e6ca8d24fd21","year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.026322Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:d981dbd4016c8988556b3897c82a9047078b46e3729cfd0d501affc86a3a9f16","observation_id":"3ba475b6-4270-4365-8c17-69f4971147bb","resolution":{"observed_at":"2026-08-15T23:45:54.227682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.207574Z","title":"Adaptive prototype learning and allocation for few-shot segmentation","venue":null,"work_id":"14c54c9e-e3be-4e76-b6ce-a6858dd6a279","year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.030624Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:1be20ac81d8f9f1b6408e4c2ca9a8595077dc694d3c7ce289221cbf9fb01aa32","observation_id":"74c4d582-3c12-489a-a629-610dfde4e711","resolution":{"observed_at":"2026-08-15T23:45:54.212236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1705.07206","last_updated":"2018-03-15T03:09:48Z","snapshot_observed_at":"2026-08-14T20:59:26.464130Z","submitted_at":"2017-05-19T21:59:09Z","title":"Multiple-Human Parsing in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.07206","snapshot_observed_at":"2026-08-15T23:45:53.035371Z","title":"Multiple- human parsing in the wild.arXiv preprint arXiv:1705.07206,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.035371Z"},"links":{"cited_paper":"/paper/1705.07206","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:48e4b91b1d53a7c8229c268aab79d169b8ff2e0dc6548bfa30d3174ae2daf4d0","observation_id":"1c3b275e-e1d0-41ce-8206-877bfec78201","resolution":{"observed_at":"2026-08-15T23:45:53.035371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.192472Z","title":"Clip surgery for better explainability with enhancement in open- vocabulary tasks.arXiv e-prints, pages arXiv–2304, 2023","venue":null,"work_id":"24c079aa-bc82-4056-a9db-f268b130aff9","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.040400Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:fe796a70900238269442987cf4ee6fed3c61034987d245e5eab1f05c18512cfc","observation_id":"8246476c-97ef-4575-8557-5904a5027230","resolution":{"observed_at":"2026-08-15T23:45:54.197681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.044661Z","title":"Learning without forgetting","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.044661Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:302a47971a9bb3ed1e6692b7d92a8cabfea5714639c53f829e5b72d21b13a39a","observation_id":"1d958cc3-dba2-4450-ba3b-52937d216dc8","resolution":{"observed_at":"2026-08-15T23:45:53.044661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.168609Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":"05925d99-f568-4b19-9e0f-e805f866a8db","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.049066Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:5b46039ef199618749dd738c919db591202a92d54d6f79bcea574e828c7c87c1","observation_id":"666d3d19-53e1-4f1d-9461-42c7691ec837","resolution":{"observed_at":"2026-08-15T23:45:54.173005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.154800Z","title":"Learning non-target knowledge for few- shot semantic segmentation","venue":null,"work_id":"119d6767-38de-464b-ba85-df8f9d5df858","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.053326Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:4502e62867c69195d0cdb28be4e9da5e841c868b95f66d3d1188df6fae652b6f","observation_id":"34844ea3-1bb0-4b8c-bb72-e1a089b2c5e1","resolution":{"observed_at":"2026-08-15T23:45:54.159232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13310","last_updated":"2024-01-19T13:03:04Z","snapshot_observed_at":"2026-08-19T23:04:38.178579Z","submitted_at":"2023-05-22T17:59:43Z","title":"Matcher: Segment Anything with One Shot Using All-Purpose Feature Matching","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13310","snapshot_observed_at":"2026-08-15T23:45:53.057371Z","title":"Matcher: Segment anything with one shot using all-purpose feature matching.arXiv preprint arXiv:2305.13310, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.057371Z"},"links":{"cited_paper":"/paper/2305.13310","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:1e787ec259831b9392a90de370a2ba49b4589d14ef113a9d42d24723af90c282","observation_id":"9cb52e1d-d651-48db-9115-d528c14678ba","resolution":{"observed_at":"2026-08-15T23:45:53.057371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.140405Z","title":"A simple im- age segmentation framework via in-context examples","venue":null,"work_id":"6866504b-4814-40ab-8ed6-0bd16c0608ab","year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.061846Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:ea32900d01bc2ef9ad438ee367df8cb360f5fb11448fad86b3625921ea7134b6","observation_id":"d878ad29-2375-4ff2-bc77-621042c4779e","resolution":{"observed_at":"2026-08-15T23:45:54.145206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.125097Z","title":"Image segmentation using text and image prompts","venue":null,"work_id":"d6025fef-9a18-49d6-87dd-def1793d4ca6","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.066095Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:1cba3a6439866e869344f311088f4517f030235676c92b8c87c7f163ef99e875","observation_id":"549f8c6b-50ac-450a-86ef-7c4b9df0f0d3","resolution":{"observed_at":"2026-08-15T23:45:54.130169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.109215Z","title":"Uavid: A semantic segmentation dataset for uav imagery.ISPRS journal of photogrammetry and remote sensing, 165:108–119, 2020","venue":null,"work_id":"bbd5c51d-f9a8-4fec-9b5d-0947a91af54f","year":2020},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.070477Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:78ea4fb2044033eddeb69f33b89a2324d766c43fa334aa344a6c3629e50e4c89","observation_id":"3624d1ab-5104-4877-80e4-3cb16da647f2","resolution":{"observed_at":"2026-08-15T23:45:54.114024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.094215Z","title":"Hypercorre- lation squeeze for few-shot segmentation","venue":null,"work_id":"4053cbcb-d9d8-472c-b310-3dff744096c3","year":2021},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.074514Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:346e2b44902dbe717b3cb2f539b42e17cffb761d46eb62914998f5f3f8f31a0b","observation_id":"5629eefe-2f0d-4a97-bdd8-6e7ce5e36b3a","resolution":{"observed_at":"2026-08-15T23:45:54.098854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.079433Z","title":"mask dataset.https://universe.roboflow","venue":null,"work_id":"a3def682-e565-4004-b5a6-af10f578b406","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.078876Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:3bb582b4a6f97b390d83b0c4efc1c7a80614f2726dd94353ccfaf94e790fa8b9","observation_id":"97f7c763-32b4-4067-b8d6-bbed627cab2a","resolution":{"observed_at":"2026-08-15T23:45:54.084217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.064203Z","title":"Open vocabulary semantic segmentation with patch aligned con- trastive learning","venue":null,"work_id":"e79175dc-49b0-4363-8494-a5f92872c3fc","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.083145Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:7675fa268fdba7184c6511167a7621074d7ecc01f8891f8203063a1a893a5990","observation_id":"71836c1e-64e7-40a1-8c27-979f1e442d8a","resolution":{"observed_at":"2026-08-15T23:45:54.068952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.11998","last_updated":"2024-12-16T17:26:06Z","snapshot_observed_at":"2026-08-15T00:25:24.205951Z","submitted_at":"2024-12-16T17:26:06Z","title":"SAMIC: Segment Anything with In-Context Spatial Prompt Engineering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.11998","snapshot_observed_at":"2026-08-15T23:45:53.087437Z","title":"Samic: Segment anything with in-context spa- tial prompt engineering.arXiv preprint arXiv:2412.11998,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.087437Z"},"links":{"cited_paper":"/paper/2412.11998","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:895cc498cd74fe1a694a0576e8298b220a40cf97cda73f586f4165e37ee29a69","observation_id":"aef46ded-96c6-4007-80a4-ffd3a2117d35","resolution":{"observed_at":"2026-08-15T23:45:53.087437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.049148Z","title":"Trash (v2).https : / / universe","venue":null,"work_id":"03e58ce7-5bc4-481d-9f0b-d6535f5a25e4","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.091918Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:93062f17482728841f5682989df97f8dbf212d355d2d07287c262bde42e4aa01","observation_id":"97a5551d-9da9-46de-a047-f66b4b5e0532","resolution":{"observed_at":"2026-08-15T23:45:54.053907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-17T13:03:40.359628Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-15T23:45:53.096097Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.096097Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:f01d3926054853bc76032c7778b1e9248a42e1fe0b98701d1a9eeaa109186a44","observation_id":"a5f680bd-7471-4949-864a-94794f9a6c06","resolution":{"observed_at":"2026-08-15T23:45:53.096097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.100519Z","title":"Continual lifelong learning with neural networks: A review.Neural networks, 113:54–71,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.100519Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:77392a5e5ceb1a2a74b7f06c0d813da6c136e7e864d0da31cfea9ff0ab16da0b","observation_id":"f1107969-8375-4a1a-bbcf-3e21fa1cfe07","resolution":{"observed_at":"2026-08-15T23:45:53.100519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.024361Z","title":"A closer look at self-training for zero-label semantic segmentation","venue":null,"work_id":"6cb05eca-8024-4b5d-9b54-042ead6ec237","year":2021},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.105802Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:d4920c9dace544e976fb509e95112238153fe5105a781d4160f3909dee1f857d","observation_id":"581c8984-c2d5-4cb8-9c73-fab07214db50","resolution":{"observed_at":"2026-08-15T23:45:54.029285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:54.009513Z","title":"Freeseg: Unified, universal and open-vocabulary im- age segmentation","venue":null,"work_id":"ba693dfe-5886-44d9-9a8d-5c6e1819d92b","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.109952Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:fb2d3534f1b9fd7137ab36eb099179ce3babf8b38055211be1932b5bebe93570","observation_id":"2b84f3b9-b925-47a0-af14-7b01c16f6502","resolution":{"observed_at":"2026-08-15T23:45:54.014182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.994442Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"d93c8cc5-4e5c-4e21-a27a-05500425228d","year":2021},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.114132Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:fd11a834dc770b55c41a413c03a4af8ea7c56de9101feef0c4282139e3da96d9","observation_id":"1a7da7fc-e7f6-4cfa-b9e0-c4e65e26cd9a","resolution":{"observed_at":"2026-08-15T23:45:53.999161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.118432Z","title":"Zero-shot text-to-image generation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.118432Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:a393ccd06f4d68c9af5916adae6e4e86e46b4e69e0d2de5a297fe269f2f3d122","observation_id":"2aa6dad9-ba2e-49ee-9391-2011536e6254","resolution":{"observed_at":"2026-08-15T23:45:53.118432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-15T23:45:53.122459Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.122459Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:2a2609cc9cb21c3abbaaa52f240470a3ea231fd6ec4fb2cb9ef9aa37d7e6121c","observation_id":"25d8415e-9ef7-4992-99da-d7ba09d72d31","resolution":{"observed_at":"2026-08-15T23:45:53.122459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.126852Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.126852Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:dbafb6882d3477a1900f46581482d8ca69613f09d560d63126d5ef64776b07fc","observation_id":"1f94a279-c0f2-42c9-8653-8f42f68f4249","resolution":{"observed_at":"2026-08-15T23:45:53.126852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1709.03410","last_updated":"2017-09-11T14:34:58Z","snapshot_observed_at":"2026-08-17T15:29:01.385951Z","submitted_at":"2017-09-11T14:34:58Z","title":"One-Shot Learning for Semantic Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1709.03410","snapshot_observed_at":"2026-08-15T23:45:53.131067Z","title":"One-shot learning for semantic segmentation","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.131067Z"},"links":{"cited_paper":"/paper/1709.03410","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:da18407c4de2879aa10943b291904270b3fdd6c217865d2d0df41189cbdab542","observation_id":"43ae1153-2c55-4443-af3c-452a70dce23b","resolution":{"observed_at":"2026-08-15T23:45:53.131067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.961136Z","title":"Reco: Re- trieve and co-segment for zero-shot transfer.NeurIPS, 35: 33754–33767, 2022","venue":null,"work_id":"4eac5838-4a64-4621-9187-f76c592db976","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.135759Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:7fb2a3ce1f3773a3f66d913cb29069be322b6473870d05c4992d2e4ac4599a8a","observation_id":"8a2cb2dd-46b9-47ea-ac97-0e713764bd90","resolution":{"observed_at":"2026-08-15T23:45:53.965974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.140307Z","title":"What does clip know about a red circle? visual prompt engineering for vlms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.140307Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:e92e60139fb70bddaae09528037ca3118ecdf8843eba72feabc7948a5447fb55","observation_id":"e5c91157-c8e8-4a2b-9e53-b939df761d18","resolution":{"observed_at":"2026-08-15T23:45:53.140307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.935651Z","title":"Vrp-sam: Sam with visual reference prompt","venue":null,"work_id":"c4663c29-513e-40b9-a0bf-83691a89ea4a","year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.144512Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:94e274965478b48a2fcd8026e8e6f862aa15323b8201db4a41f1aace2dc25000","observation_id":"0d5d9d12-b5d7-4cc8-b8a6-9eb6bbfa359f","resolution":{"observed_at":"2026-08-15T23:45:53.940864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.920004Z","title":"abc dataset.https : / / universe","venue":null,"work_id":"02607b72-1949-4edb-9785-c3df2fc4b9e7","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.148953Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:5d0acf637f3f6f97c083f085738429927e819b62a9823b4fef8c5ff1ff0eaef0","observation_id":"41dea74e-13af-4fc1-b791-cedf3288e918","resolution":{"observed_at":"2026-08-15T23:45:53.925189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.903962Z","title":"Springer, 1998","venue":null,"work_id":"ee8cf0a2-7406-4c60-8cdd-ce0380770452","year":1998},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.153269Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:b60c3d92ee49b7610a1c1a9c91d6c0e37cf05a21474d0d090f6e7b10a6978f52","observation_id":"0e4ee27d-78ec-4011-a228-56d31b5c171d","resolution":{"observed_at":"2026-08-15T23:45:53.909431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.157591Z","title":"Sclip: Rethinking self-attention for dense vision-language inference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.157591Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:a82eda70316f208b496e92bec7dc7045d19e6e4615046f5d9b49fc62baaef68a","observation_id":"5ef639c4-2f0e-41a7-b06e-a5ee12f253ee","resolution":{"observed_at":"2026-08-15T23:45:53.157591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.773478Z","title":"Few-shot semantic seg- mentation with democratic attention networks","venue":null,"work_id":"36396ad1-91d7-4ccc-8a4d-d91daf1241c6","year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.162059Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:28245e302a823e5d1cbb4d687d6ef81c2914c65b8584025cf5bf24e75baa1a98","observation_id":"3abf3147-d918-49ab-ab98-2ce66c686ce1","resolution":{"observed_at":"2026-08-15T23:45:53.777981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.759424Z","title":"Sam-clip: Merging vision foundation models towards semantic and spatial understanding","venue":null,"work_id":"da0b39f1-8140-4cb1-b601-48cc1ce6f5fe","year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.166415Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:f8c2d1845bf97f13f875ad1de94e224fdc01db1c64d40c8614f13cbdc8736f18","observation_id":"6de3229a-ed07-4cfc-88c6-825745400d0c","resolution":{"observed_at":"2026-08-15T23:45:53.763810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.170470Z","title":"Loveda: A remote sensing land-cover dataset for domain adaptive semantic segmentation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.170470Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:0b7cfa32859c774b71db1fe561fc555a7611a53a177ef4277a1e3f0e2461ab01","observation_id":"04a3d8c1-92aa-445f-b542-436a4bb97f5b","resolution":{"observed_at":"2026-08-15T23:45:53.170470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.734563Z","title":"Review of large vision models and visual prompt engineering.Meta-Radiology, 1(3):100047,","venue":null,"work_id":"a8e027a7-1e11-44fd-9e1b-dc5628d491b9","year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.174911Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:a1c71bc790e0b7fcc5c759e148aea9c5f1d654bfc285a5d2ff2c7959b946124a","observation_id":"122cd54f-83a1-44b2-a788-15d5379217e8","resolution":{"observed_at":"2026-08-15T23:45:53.739252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.720426Z","title":"Panet: Few-shot image semantic segmenta- tion with prototype alignment","venue":null,"work_id":"ebe13896-353e-4f99-8499-23809f743691","year":2019},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.179043Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:92c371d3da1546076eea7470eb4e6761ca76d72d505d23aca20f9c287fc7b105","observation_id":"e3f1bff4-2442-4be1-95bc-0856142408e7","resolution":{"observed_at":"2026-08-15T23:45:53.724911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.705612Z","title":"Images speak in images: A generalist painter for in-context visual learning","venue":null,"work_id":"ade52339-4693-4d12-80d0-436ea8afb82a","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.183285Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:9c6181765fe994d82dd8f8c9d07a9aca2fd37e127d55c5f07c99d2962ed79d61","observation_id":"9cd560e3-ad50-4481-9c31-1caef4044439","resolution":{"observed_at":"2026-08-15T23:45:53.710265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.690825Z","title":"Seggpt: Towards seg- menting everything in context","venue":null,"work_id":"d3bd926d-6275-4ab2-a2fe-2a51b4724d3d","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.187482Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:375c6b256fad1e151b4bbcfc88e83a17ce2c7c9c0a6b8fdeb3f78f71347fbbf8","observation_id":"a9e7135b-cfba-483a-93d5-f771af26e96e","resolution":{"observed_at":"2026-08-15T23:45:53.695800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.675460Z","title":"Clip-dinoiser: Teaching clip a few dino tricks for open- vocabulary semantic segmentation","venue":null,"work_id":"1cb525a9-5cce-4a07-9932-bff9724f4541","year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.191659Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:95667db3ee5ef0f32f808dbdf72ac57df8ce6b3d31248e61e4182833bcb35a59","observation_id":"5cd2d9b5-81c7-4a44-b581-b7bd0d30f92f","resolution":{"observed_at":"2026-08-15T23:45:53.680376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.660896Z","title":"Semantic projection network for zero-and few-label semantic segmentation","venue":null,"work_id":"1bda554d-16bc-468f-9375-2bb3655275e1","year":2019},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.195947Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:33aa1c78ff293d603a446c62fcb9dc2151d717940611f7f17d9be29b7bc52bab","observation_id":"21c1811d-41d2-40a0-a02e-b706920559a8","resolution":{"observed_at":"2026-08-15T23:45:53.665565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.645772Z","title":"Cat-sam: Con- ditional tuning for few-shot adaptation of segment anything model","venue":null,"work_id":"20b0f048-df51-4058-beb9-4816ff2615ee","year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.200304Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:e695938e464c53552a242320e205bd5eff5a91439d14ec92d25779d9956a3c84","observation_id":"b94b039f-64cb-491c-982f-749e0e9479b4","resolution":{"observed_at":"2026-08-15T23:45:53.650932Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.630317Z","title":"piiz dataset.https://universe.roboflow.com/ y-rgb4q/piiz, 2023","venue":null,"work_id":"4b5c7579-2455-492e-ae74-72e4d9bf2e1a","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.204590Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:781e291e6e6568a9b3c4ba154299132c3a455c6144aff6db664f0b3f5d6dad44","observation_id":"85b6580a-bb42-4cfc-a07d-792b869b0023","resolution":{"observed_at":"2026-08-15T23:45:53.635361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.616278Z","title":"Contin- ual learning through synaptic intelligence","venue":null,"work_id":"1b26a3d9-08bb-4ba6-ab28-8ad7a7470027","year":2017},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.209125Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:539538cfdfa0ae2c487fbd1f8350016232a379ba2e16b4b16d2170b9e4d1123a","observation_id":"3bbe566d-2640-4382-a278-9c798632d8cd","resolution":{"observed_at":"2026-08-15T23:45:53.621040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.601580Z","title":"Bridge the points: Graph-based few-shot segment anything semantically.NeurIPS, 37:33232–33261, 2024","venue":null,"work_id":"f1e1ffff-ce17-41c5-9fd0-822a04ae216c","year":2024},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.213370Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:42eb579a4c98add8c030d08304c8af97420fb2a4f16bc0908e0d99014e0e68cb","observation_id":"f00ecdaf-63f4-4163-87ea-4fa2721f3a2d","resolution":{"observed_at":"2026-08-15T23:45:53.606550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.587072Z","title":"Pyramid graph networks with connection attentions for region-based one-shot semantic segmentation","venue":null,"work_id":"4f9fd200-a656-48d8-b898-1e942acb75ec","year":2019},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.217839Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:ec118ea12311248e78b26ead9c9facaccef390ac26fe8a8945825e17de00e30d","observation_id":"c69317a7-63c7-4d72-8844-babfe4fd5bac","resolution":{"observed_at":"2026-08-15T23:45:53.591720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.571342Z","title":"Few-shot segmentation via cycle-consistent trans- former.NeurIPS, 34:21984–21996, 2021","venue":null,"work_id":"bed0679a-6eeb-4a91-90e4-748c266107e2","year":2021},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.222951Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:429494655cd7b8cb3be3508c07ec170f1027eee33ac95b0026209ab2964d1347","observation_id":"0fd5bbf9-647c-4228-b5b0-909e31b42e8e","resolution":{"observed_at":"2026-08-15T23:45:53.577198Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.555791Z","title":"Pidray: A large-scale x-ray benchmark for real-world prohibited item detection.International Journal of Computer Vision, 131 (12):3170–3192, 2023","venue":null,"work_id":"9f1457b1-7683-4cb0-a238-77f9254e0e6b","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.227410Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:806dc7aff13c910f7f7008277bff0e3886e6d2109e81158ce8dd96c6ec02a279","observation_id":"6ba3fe44-ec12-4c7d-b864-e396104136a0","resolution":{"observed_at":"2026-08-15T23:45:53.561329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.03048","last_updated":"2023-10-04T01:15:21Z","snapshot_observed_at":"2026-08-16T15:35:32.915462Z","submitted_at":"2023-05-04T17:59:36Z","title":"Personalize Segment Anything Model with One Shot","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.03048","snapshot_observed_at":"2026-08-15T23:45:53.231801Z","title":"Personalize segment anything model with one shot","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.231801Z"},"links":{"cited_paper":"/paper/2305.03048","citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:84244c28402823b88903312489183953c0ad6ea0afd628bbb77265c83ca7508f","observation_id":"f6287afd-ec0b-4319-b559-2fc287983d07","resolution":{"observed_at":"2026-08-15T23:45:53.231801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.541191Z","title":"Scene parsing through ade20k dataset","venue":null,"work_id":"1478c5c5-0245-45b7-a6a4-fa8ee69341f2","year":2017},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.236292Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:7e6422a8e2ca95395aeffffd829deb3a3d0416afa979cd36e101507839c0f196","observation_id":"ad0e7724-d552-4e0e-9caa-c93ddb730b97","resolution":{"observed_at":"2026-08-15T23:45:53.545837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.525181Z","title":"Extract free dense labels from clip","venue":null,"work_id":"82cb2820-1302-478a-a552-76232db14945","year":2022},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.240855Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:18839fab6a21cbc6583aac990fb2356bb2c04b5167b503c6f5fd929064e32c4c","observation_id":"a1daea86-e065-451e-8409-0fa59483d560","resolution":{"observed_at":"2026-08-15T23:45:53.529975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.509790Z","title":"Generalized decoding for pixel, image, and language","venue":null,"work_id":"8c1f519c-ee58-40d9-ba93-011b0ff3e395","year":2023},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.245644Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:bb78c8e86f03c44e4da68e40d6336fa70617da4c560662b0c32d9b459f2bf66f","observation_id":"739e725d-aec8-40ad-95f5-45c03c978df1","resolution":{"observed_at":"2026-08-15T23:45:53.514793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.494799Z","title":null,"venue":null,"work_id":"4190deaa-9cc1-458a-91e8-d3d02401ae83","year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.249926Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:f425b6e890f2839ad8edac4cec141b2da557dfa7e9d68f30be66bfaf497628c6","observation_id":"fda02608-a3cd-4ebc-a3f1-1dd456f00b19","resolution":{"observed_at":"2026-08-15T23:45:53.499588Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.478921Z","title":"Open-vocabulary methods.To ensure a fair comparison, we report the results for open-vocabulary methods without applying any mask refinement step (e.g","venue":null,"work_id":"3d47beed-db85-4717-b75f-46d465ce2f1f","year":null},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.254178Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:c2ab94643f00a452e3e1b74a1e0f051f9c0f199a8b591608c1797e049b256a1e","observation_id":"fe53199c-e760-42ae-a120-a90201a346ce","resolution":{"observed_at":"2026-08-15T23:45:53.483961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T23:45:53.463103Z","title":"ADE20KThe ADE20K [75] dataset is made up of 150 classes","venue":null,"work_id":"f68de8f3-fbc8-4710-81b0-08a13bd6a85e","year":2012},"citing_paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-15T23:45:53.258575Z"},"links":{"citing_paper":"/paper/2505.06280"},"observation_digest":"sha256:d74589e5d1d4490e5ab540703f2b3789de1d7ef8dd45b1c4441a38ad26dc4283","observation_id":"6cce04f5-8251-4902-85b4-7fff1216dc52","resolution":{"observed_at":"2026-08-15T23:45:53.468004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.06280","last_updated":"2025-05-06T20:15:30Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-18T18:06:51.076168Z","submitted_at":"2025-05-06T20:15:30Z","title":"Show or Tell? A Benchmark To Evaluate Visual and Textual Prompts in Semantic Segmentation"},"reference_resolution":{"displayed":80,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":1,"verified_fuzzy":59},"total_outbound_references":80},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 80 of 80 outbound references and 0 inbound Pith citation observations for arXiv:2505.06280."}