{"as_of":"2026-08-18T16:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7f7a309b960451eeca11cf61a640ae05af10f3e01b2a533ec2f827a48fbb4338","coverage":[{"denominator":71,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":71,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T18:02:15.217453Z","state":"measured"},{"denominator":73,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":73,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-21T07:43:28.627414Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T07:44:02.855304Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"cited_work":{"arxiv_id":"2411.12044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.12044","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://arxiv.org/abs/2411.12044","venue":null,"work_id":"c86637e9-4b15-4205-9dbb-eb7dcb36bc4e","year":2024},"citing_paper":{"arxiv_id":"2605.17630","last_updated":"2026-05-19T18:36:14Z","snapshot_observed_at":"2026-08-14T23:43:56.106415Z","submitted_at":"2026-05-17T19:51:32Z","title":"SegRAG: Training-Free Retrieval-Augmented Semantic Segmentation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-20T13:51:35.769341Z"},"links":{"cited_paper":"/paper/2411.12044","citing_paper":"/paper/2605.17630"},"observation_digest":"sha256:d94c8a4493bde6c4169cad2bcc8e80ebfb8258bfe43f2eda53c0a8d23a2d62bd","observation_id":"5b4f4557-10d4-48b9-b2a5-b444dc26a802","resolution":{"observed_at":"2026-05-20T13:53:19.909663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"cited_work":{"arxiv_id":"2411.12044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.12044","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://arxiv.org/abs/2411.12044","venue":null,"work_id":"c86637e9-4b15-4205-9dbb-eb7dcb36bc4e","year":2024},"citing_paper":{"arxiv_id":"2605.17630","last_updated":"2026-05-19T18:36:14Z","snapshot_observed_at":"2026-08-14T23:43:56.106415Z","submitted_at":"2026-05-17T19:51:32Z","title":"SegRAG: Training-Free Retrieval-Augmented Semantic Segmentation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-21T07:43:28.627414Z"},"links":{"cited_paper":"/paper/2411.12044","citing_paper":"/paper/2605.17630"},"observation_digest":"sha256:45d8ea4859b140a79d76ad28f04213f0faed36a410429894338ceb75bdf1baa2","observation_id":"a17b1e41-9266-4613-a162-a90dc371c1a4","resolution":{"observed_at":"2026-05-21T07:44:02.856656Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.12044/citation-record","integrity":"/paper/2411.12044/integrity","json":"/paper/2411.12044/citation-record.json","paper":"/paper/2411.12044"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.260424Z","title":"Cdul: Clip-driven unsupervised learning for multi-label image classification","venue":null,"work_id":"480e97d5-dba8-4561-8fd1-e107b85ef68c","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.881814Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:265b6a30413bb8f6c4812e86fe4660892ec2ef57a3fbda1036b5b80834a814de","observation_id":"211d8a3e-c8e6-4690-8246-333147dd6420","resolution":{"observed_at":"2026-08-12T18:02:16.266548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:14.887139Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.887139Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:aefad909199435e0df2a79c89a483f7041411aefd63b4412c880957ba474317a","observation_id":"b6ea9513-2029-4441-8569-f4895551b810","resolution":{"observed_at":"2026-08-12T18:02:14.887139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.228214Z","title":"Self- supervised multimodal versatile networks","venue":null,"work_id":"2353e9bf-3812-4df3-85bd-16289031c1cc","year":2020},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.891219Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:174936edc65327a53368c562d8856017f32833b43eb242508fd97b6a146afc3d","observation_id":"72f9f8ae-5d9f-4dc8-8eb9-a694084cd171","resolution":{"observed_at":"2026-08-12T18:02:16.240261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:14.896060Z","title":"Single-stage semantic segmentation from image labels","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.896060Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:ce3ced8bdf84ca0387a6548bd2e706a034a3e3916f43be1f5a61517e1d43653c","observation_id":"da910b16-5db1-4bf5-a5b7-051b5d23c11a","resolution":{"observed_at":"2026-08-12T18:02:14.896060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.205358Z","title":"Fossil: Free open-vocabulary semantic seg- mentation through synthetic references retrieval","venue":null,"work_id":"631786c2-e4bf-4b6c-947a-cde4ba479bb2","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.900159Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:a4c76c0f366a84cfaef22bdbb4f43f1072378e3f2530a895ce9fdb077b2367d3","observation_id":"481c2f86-fb96-4dae-b2d3-1901bb82c67d","resolution":{"observed_at":"2026-08-12T18:02:16.210189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.190380Z","title":"Grounding everything: Emerging localiza- tion properties in vision-language transformers","venue":null,"work_id":"62afa050-e1bd-488c-b26d-56b925d99ecd","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.904567Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:0d465e9121d4f690da365b4f049a89374851a6b0e58fb384beb6c37c60d2447b","observation_id":"dff6d89b-1641-4696-843d-b213366b72ab","resolution":{"observed_at":"2026-08-12T18:02:16.195655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.175390Z","title":"Brown, Benjamin Mann, Nick Ryder, Melanie Sub- biah, Jared Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al","venue":null,"work_id":"e8960250-b5e2-4b1c-9645-4a3adf715738","year":2020},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.908491Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:74354ddc36935d67f31c702fb713e1c40c9bba2c39af5762ec46b364991cdb7c","observation_id":"6fb569d3-48eb-4022-a872-9aeb30cb984f","resolution":{"observed_at":"2026-08-12T18:02:16.180194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.162299Z","title":"Coco- stuff: Thing and stuff classes in context","venue":null,"work_id":"dd1f8ad9-5b53-4a09-a956-57ee08ff2724","year":2018},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.912479Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:eab19123b5979e907d6d59e891916a683041ec8e6d91a131272c64eec9904242","observation_id":"4ab646cf-a78e-4811-8387-bf3a92a7999e","resolution":{"observed_at":"2026-08-12T18:02:16.166805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.147898Z","title":"Cambridge dictionary","venue":null,"work_id":"77222923-c9ba-431e-850a-8d7a8d64f119","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.916571Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:d9bad7d2ac351f830b57ff0fe33335a0e2b16e3d44dabb9ad3929df42d710cd2","observation_id":"3e997ae8-6fdf-449d-81fa-949b1d6e4a11","resolution":{"observed_at":"2026-08-12T18:02:16.153216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.136647Z","title":"Learn- ing to generate text-grounded mask for open-world semantic segmentation from only image-text pairs","venue":null,"work_id":"b2dea88c-cb8a-4712-be61-9eddcf633530","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.920494Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:ec10aa32f528b738b135bee24ec89f3d8c024ee81c11d38d840550ee90d1669a","observation_id":"6f44682f-14fd-47f9-b54a-a974ef599626","resolution":{"observed_at":"2026-08-12T18:02:16.140467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:14.924613Z","title":"Reproducible scal- ing laws for contrastive language-image learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.924613Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:a81d6d37fffed66be19370fca026836ddf95dbb25e27c90f56ad937dadc2f989","observation_id":"8c13f81f-b621-4f41-8757-12e61577acf3","resolution":{"observed_at":"2026-08-12T18:02:14.924613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.115344Z","title":"The cityscapes dataset for semantic urban scene understanding","venue":null,"work_id":"581aaf17-8b46-4fc7-8367-581fa328e96d","year":2016},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.928945Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:080b6ece18f2f41a09037762fe985b0bef7e73d97b4f9d98f0e85998f6d77000","observation_id":"c0887d83-5411-474e-9f86-d183190c6a44","resolution":{"observed_at":"2026-08-12T18:02:16.119894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:14.934430Z","title":"Imagenet: A large-scale hierarchical image database","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.934430Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:03744733a96939caedff17c8fdc58e92a0fced8778022ccb9058a6ad9e43c57d","observation_id":"0f1cf99a-31f4-4c16-8f8c-dc718bbd7c6d","resolution":{"observed_at":"2026-08-12T18:02:14.934430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.04155","last_updated":"2023-04-09T04:06:59Z","snapshot_observed_at":"2026-08-16T15:41:47.817321Z","submitted_at":"2023-04-09T04:06:59Z","title":"Segment Anything Model (SAM) for Digital Pathology: Assess Zero-shot Segmentation on Whole Slide Imaging","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.04155","snapshot_observed_at":"2026-08-12T18:02:14.939125Z","title":"Segment anything model (sam) for digital pathology: Assess zero- shot segmentation on whole slide imaging","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.939125Z"},"links":{"cited_paper":"/paper/2304.04155","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:873d29c0b9563a563a4efa6ee21b34825350fd2b8a7454e169fb5095975ade95","observation_id":"c21f8ba0-67ec-4ace-a42f-393c5e3a02af","resolution":{"observed_at":"2026-08-12T18:02:14.939125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-08-14T18:16:28.847993Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-12T18:02:14.944316Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.944316Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:3ad19821a6d462ac80fb0b5551da297d1b9ab3189e3a72e3c2a56cbc3b846d88","observation_id":"0d0a5f28-718a-46ba-b984-183d0417ccee","resolution":{"observed_at":"2026-08-12T18:02:14.944316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:14.950514Z","title":"De- coupling zero-shot semantic segmentation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.950514Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:3749d1d0bfc891b86caeaeeda26b6a02e6df3904463ae90e49f6c09092abd66c","observation_id":"3409ab2c-b2b2-478a-9d24-8000c59fcfa1","resolution":{"observed_at":"2026-08-12T18:02:14.950514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.086995Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":"c1eefbe4-f23e-4112-a765-90e6f121f28c","year":2020},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.954949Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:e1b58b44ee01c9b829e0928c5193e1d8339fba3dff72703c2436be78b8919d6a","observation_id":"29ad03de-4071-454d-8fe6-17bcb00fcf8c","resolution":{"observed_at":"2026-08-12T18:02:16.091190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.073646Z","title":"Learning to prompt for open-vocabulary ob- ject detection with vision-language model","venue":null,"work_id":"ecbd856a-7c92-46d5-8fe2-7b83131f23cc","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.960267Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:bed4b15a94c7df81070f006bc856d8c4909a1a657fa21ca9e935e46aed6e48b9","observation_id":"b466df33-5e4c-481e-923b-266465469f36","resolution":{"observed_at":"2026-08-12T18:02:16.078230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-12T18:02:14.965778Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.965778Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:df02f8df19665eb3ae0b63e8b1f17155f9651b713d646b539bd3911e9c0683f6","observation_id":"47293814-2e61-4607-9305-b2430122f3db","resolution":{"observed_at":"2026-08-12T18:02:14.965778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.060572Z","title":"The pascal visual object classes (voc) challenge","venue":null,"work_id":"c036bdb7-cc6e-40a7-86d2-e1ae106a402c","year":2010},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.970745Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:c550fce74374d6cedbb81b654ae94c4d7c14af33ffd27a6844a407140721cfd0","observation_id":"5c5c00ca-f4a6-45cf-96db-e7116c81cc76","resolution":{"observed_at":"2026-08-12T18:02:16.064978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.048230Z","title":"Improving clip training with language rewrites","venue":null,"work_id":"21aea9ec-28d8-4697-bdff-3065d5f1b409","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.975042Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:0a89e888227ad269876db6c284b17db96547a929741dc92ad98363c1888dc195","observation_id":"0c34932b-d34d-43fd-922d-914cb46083a6","resolution":{"observed_at":"2026-08-12T18:02:16.052158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.034759Z","title":"In- terpreting clip’s image representation via text-based decom- position","venue":null,"work_id":"938f6978-3569-489f-aca4-0029bd93f0af","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.980372Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:e939343025fe2986df94348c59c710fe53463f93290cf8e47bd47735343798d1","observation_id":"b510a98d-7f1c-42b4-aa19-6017674f0854","resolution":{"observed_at":"2026-08-12T18:02:16.039663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.020995Z","title":"Scal- ing open-vocabulary image segmentation with image-level labels","venue":null,"work_id":"9dd62b2d-8ad3-4b7f-a991-7f1d46b34687","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.984804Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:871a641e8132da484ae25360d063cd063c943f37cc351b25ddeec6c7acf7d839","observation_id":"5dd071c2-6165-4e5d-ba25-5eca120f0394","resolution":{"observed_at":"2026-08-12T18:02:16.025991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:16.009476Z","title":"Open-vocabulary object detection via vision and language knowledge distillation","venue":null,"work_id":"5e8449fe-2e2d-473a-9e6a-857e983e2193","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.992137Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:d2f996758541aca0b588359f92aba0ee50f21ca40dff463a9dffe3e4c87a290f","observation_id":"9511a246-35da-4c60-9430-440bd34b54b8","resolution":{"observed_at":"2026-08-12T18:02:16.013414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08181","last_updated":"2024-09-16T20:11:48Z","snapshot_observed_at":"2026-08-16T14:01:33.981371Z","submitted_at":"2024-04-12T01:08:04Z","title":"Pay Attention to Your Neighbours: Training-Free Open-Vocabulary Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08181","snapshot_observed_at":"2026-08-12T18:02:14.998915Z","title":"Pay attention to your neighbours: Training-free open-vocabulary semantic segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:14.998915Z"},"links":{"cited_paper":"/paper/2404.08181","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:55fbdc84b0e456911d07d5acadda2b991087e6184fe27a32f7154fc08b829c88","observation_id":"1582c321-d54b-4cc8-bee0-18cbb66c0b0b","resolution":{"observed_at":"2026-08-12T18:02:14.998915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.003687Z","title":"Open-vocabulary semantic segmentation with decou- pled one-pass network","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.003687Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:2ce3fdfd66f675940bbe1879c5bd442bb05be70134172b28940e23273aa398e4","observation_id":"dd8ae3d4-9d8f-4ce7-b917-34f80779b8b5","resolution":{"observed_at":"2026-08-12T18:02:15.003687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.007900Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.007900Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:8f56eca83f5e662e95c99f628ae43565207aa4c45b42be1dc64c7fc49248786c","observation_id":"56431c2d-b044-4d6f-809e-53cdbf89ed86","resolution":{"observed_at":"2026-08-12T18:02:15.007900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09324","last_updated":"2023-05-05T18:58:52Z","snapshot_observed_at":"2026-08-17T14:25:32.657046Z","submitted_at":"2023-04-18T22:16:49Z","title":"Computer-Vision Benchmark Segment-Anything Model (SAM) in Medical Images: Accuracy in 12 Datasets","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09324","snapshot_observed_at":"2026-08-12T18:02:15.013156Z","title":"Computer- vision benchmark segment-anything model (sam) in med- ical images: Accuracy in 12 datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.013156Z"},"links":{"cited_paper":"/paper/2304.09324","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:ee424a2d236b61bc02bf2c9f81674634a6c1c34a3c9101147983f98c2b37da38","observation_id":"5e21e778-6fda-49d8-8017-7709597aa0b9","resolution":{"observed_at":"2026-08-12T18:02:15.013156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.982287Z","title":"Open- clip, July 2021","venue":null,"work_id":"5267b62a-a47a-440b-85b8-ec071f271645","year":2021},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.019669Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:1ca4c6e996a6e2f1ae4c045172290311188aedc01f5ad1e9ba29f41cb11af334","observation_id":"b13db7ec-998b-47b3-abe3-85d987c34028","resolution":{"observed_at":"2026-08-12T18:02:15.985978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.024874Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.024874Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:3dfef2a5d5b34402ff68da4fc69eb6683c301e24953fe3d0a9cbaaa5f0791f0e","observation_id":"02084594-29a3-4982-adc4-6e61b7efee96","resolution":{"observed_at":"2026-08-12T18:02:15.024874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.029460Z","title":"Learning mask-aware clip representations for zero-shot segmentation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.029460Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:390d4ab053306b3ee1b73682cf18281358793ab0ecb8140334c109c06548af5b","observation_id":"9e94885b-94b4-46f2-86df-bc7dcf5525de","resolution":{"observed_at":"2026-08-12T18:02:15.029460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.034305Z","title":"Segment any- thing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.034305Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:593b0d45ecd698b5cde754299477bf85f14ad79f52140621671c1b7ffbcd3754","observation_id":"f219d208-1b85-40fc-9e8c-4d9c017d0510","resolution":{"observed_at":"2026-08-12T18:02:15.034305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12442","last_updated":"2024-07-17T09:52:20Z","snapshot_observed_at":"2026-08-16T13:33:29.585467Z","submitted_at":"2024-07-17T09:52:20Z","title":"ClearCLIP: Decomposing CLIP Representations for Dense Vision-Language Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12442","snapshot_observed_at":"2026-08-12T18:02:15.038113Z","title":"Clearclip: Decom- posing clip representations for dense vision-language infer- ence","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.038113Z"},"links":{"cited_paper":"/paper/2407.12442","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:e0b9f5b08c121353f8e86f79d67604ec28a115354d1e217c2bea4d1d826a03ca","observation_id":"56afe800-ac58-4364-94b7-1eaecdf3a1e8","resolution":{"observed_at":"2026-08-12T18:02:15.038113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.042655Z","title":"Language-driven semantic seg- mentation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.042655Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:0443587088f86c842365289ee13e285e1e97513302d3e1226673ff5970418664","observation_id":"8bf95457-bb15-4e24-9c72-1bb2e9de176d","resolution":{"observed_at":"2026-08-12T18:02:15.042655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.046526Z","title":"Align before fuse: Vision and language representation learn- ing with momentum distillation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.046526Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:cdf9d8f50e48aa6fa9898be8d215de32163f8cf2f8289857ac4444593a2816c6","observation_id":"01364582-e267-4d1b-8113-0e67a3c6847f","resolution":{"observed_at":"2026-08-12T18:02:15.046526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.928132Z","title":"Cascade-clip: Cascaded vision-language embeddings alignment for zero-shot semantic segmentation","venue":null,"work_id":"3defa509-3611-49a7-91d2-9709965cf3b7","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.051512Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:bae7197d561ac8137a5c26ba4eb665310641a7bb76189a5365df664fd672f564","observation_id":"0b122750-5ae1-4ae5-a390-22d928c5f2cd","resolution":{"observed_at":"2026-08-12T18:02:15.933345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05653","last_updated":"2024-09-16T09:10:00Z","snapshot_observed_at":"2026-08-16T15:41:05.975016Z","submitted_at":"2023-04-12T07:16:55Z","title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05653","snapshot_observed_at":"2026-08-12T18:02:15.055788Z","title":"Clip surgery for better explainability with enhancement in open- vocabulary tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.055788Z"},"links":{"cited_paper":"/paper/2304.05653","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:883caaaaec27ea1e6ff4dd2bcfb97317153ded65ac9fbbee12b6f29ad6cc2c79","observation_id":"50d91a68-bd76-4a2a-b76a-0a60eca8e541","resolution":{"observed_at":"2026-08-12T18:02:15.055788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.060339Z","title":"Open-vocabulary object segmentation with diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.060339Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:345792b4abc51935dcee54865009e10bc27dbe72f25c06116b05d8622e81f394","observation_id":"46bf1e19-e2f0-4b83-8ed1-b30ae0c03551","resolution":{"observed_at":"2026-08-12T18:02:15.060339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.065044Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.065044Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:dad073afde71a8d86dfdac2f040fd123458fd49a5ca99c45a6ab1de72360a413","observation_id":"f59aa609-d099-4e85-ae70-8469dfecd596","resolution":{"observed_at":"2026-08-12T18:02:15.065044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.890946Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"da892a4b-9212-4ad1-9efd-a606aee51481","year":2014},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.070556Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:bb196f60ab29d3466f882d13690566515ff577afe0aa1474f45e77e9b5ed962a","observation_id":"75bcf5e0-0dff-4ecf-976b-fb57d54dcc44","resolution":{"observed_at":"2026-08-12T18:02:15.896916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.877220Z","title":"Clip is also an ef- ficient segmenter: A text-driven approach for weakly super- vised semantic segmentation","venue":null,"work_id":"6ff582f9-1fc8-4bfe-ab7f-f1b71f7f9997","year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.075174Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:0b3c02bd16c70d6d1c31c9c4fdb8fbbb48db9089c1fe61e841b63030f6ff5183","observation_id":"190940b6-a7b6-4443-9446-4fbbdb569e31","resolution":{"observed_at":"2026-08-12T18:02:15.882322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.859210Z","title":"Tagclip: A local-to-global framework to enhance open-vocabulary multi-label classification of clip without training","venue":null,"work_id":"6423bc64-9424-4f21-b95a-7dfdaa07c0a4","year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.080381Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:ae80b1f922d3bdd44fdf0019fbbf7e13faa0da74052ac8968d88ceb837a01255","observation_id":"b42af303-fa76-4e58-84bd-564ea7e94149","resolution":{"observed_at":"2026-08-12T18:02:15.864645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.843963Z","title":"Object- centric learning with slot attention","venue":null,"work_id":"9907181d-0eec-4514-9783-76ad6633d363","year":2020},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.085665Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:75c50cbe251a6a35814df9e98310ae83f0215f82a63fcae916896c71ead4ddf7","observation_id":"2d462619-d481-4064-ba83-a4e310900b11","resolution":{"observed_at":"2026-08-12T18:02:15.848421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.813821Z","title":"Segclip: Patch aggregation with learn- able centers for open-vocabulary semantic segmentation","venue":null,"work_id":"ffd9d0fe-aa0d-42a7-a66e-d4e09f97313e","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.090023Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:6b59b852968bfdced58bb5863afb4f41e3cd2821cd6ac7a9b3adbd17414b3a0f","observation_id":"b0e5cd08-77fd-4df4-9e12-04d6a2833d01","resolution":{"observed_at":"2026-08-12T18:02:15.824852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.095233Z","title":"Segment anything model for medical image analysis: an experimental study","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.095233Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:385eaa82d84154d5ae1010e55551651ad147e77f5313442e96eb5ec1a966bc1a","observation_id":"964ee126-98fa-4a52-8cad-db981c1d328d","resolution":{"observed_at":"2026-08-12T18:02:15.095233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.785037Z","title":"MetaICL: Learning to learn in context","venue":null,"work_id":"2c5f693a-ba1f-444f-ab5b-7152f35924b5","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.099576Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:a2bcef8a78a8aaa669a121043fa2b8d7183200eabc4c207092c9f216e507c783","observation_id":"6fd68484-45b2-43c2-a4ac-1d7f1d998269","resolution":{"observed_at":"2026-08-12T18:02:15.793596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.104212Z","title":"The role of context for object detection and se- mantic segmentation in the wild","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.104212Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:cd7c3ad025b5d04f73b62f3a6d623f319b4f9b9f2340e48ef5c68cf12e5509c9","observation_id":"db2a48f3-b942-4f89-96ad-20c99eda0f74","resolution":{"observed_at":"2026-08-12T18:02:15.104212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.761893Z","title":"Peters, Mark Neumann, Mohit Iyyer, Matt Gard- ner, Christopher Clark, Kenton Lee, and Luke Zettlemoyer","venue":null,"work_id":"4f0737d1-4755-42f9-8c1b-d7427935fdb7","year":2018},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.108896Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:f0a3fc1a6394442414b92ad5378f235a7caeab35137a214d952084b149d5d4d7","observation_id":"28814d13-7c08-4ade-90a9-27330caac1a0","resolution":{"observed_at":"2026-08-12T18:02:15.767280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.747569Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"e2b0fa69-d8e4-4718-9c9f-9ce3aff53436","year":2021},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.113156Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:f201cd67076786181459f3954fa361b1a0ca6ff3253fc050effa16335897c26c","observation_id":"b772ee8e-83cb-4769-aeb0-9dc5fd81e4de","resolution":{"observed_at":"2026-08-12T18:02:15.752229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.734593Z","title":"Improving language understanding by gen- erative pre-training","venue":null,"work_id":"2a56dc70-7ec5-4267-a5a5-2579772f79cb","year":2018},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.117918Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:e0313a5d8a1d4f53b5dd86fb3092fe019179da424b42452826270dca38fb5b9e","observation_id":"13723cda-a859-4258-b68d-13e5289e6123","resolution":{"observed_at":"2026-08-12T18:02:15.739310Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.123028Z","title":"Exploring the limits of transfer learning with a unified text-to-text transformer","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.123028Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:433b6ad983ff0e952b2c26e906a097a5d5bc911c6ae9758361fe8a3546912871","observation_id":"698491c3-2181-491b-b6fc-b0275cf89c4e","resolution":{"observed_at":"2026-08-12T18:02:15.123028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.710563Z","title":"Per- ceptual grouping in contrastive vision-language models","venue":null,"work_id":"35ce9a02-88c1-4067-96f5-f9d0450673f9","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.127248Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:271373c589be914eb1c01f771e2d9b16f41da16c308e6d32b630024db9e9b043","observation_id":"5808c737-4193-4742-80be-7f5637b200c0","resolution":{"observed_at":"2026-08-12T18:02:15.715493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.691746Z","title":"Laion-5b: An open large-scale dataset for train- ing next generation image-text models","venue":null,"work_id":"dd4bd0c9-224e-499c-86b9-b012515404d5","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.131814Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:95bc23073a9911958ea3906cbe9bf46b5f48b23fd960708b23173fbfa30d0847","observation_id":"a5c0e07a-e3ef-4355-aa60-abe705c1ca44","resolution":{"observed_at":"2026-08-12T18:02:15.697954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.674743Z","title":"Ex- plore the potential of clip for training-free open vocabulary semantic segmentation","venue":null,"work_id":"08b17100-c99d-4070-b0ad-26771a33f23f","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.136424Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:bd285a31f7fcc5e3bdc1c53396e9f91717e5d55cfe407940bf9f61d00db0b0e1","observation_id":"478430dc-a978-417d-b010-d8b83f948698","resolution":{"observed_at":"2026-08-12T18:02:15.680188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.658646Z","title":"Edadet: Open-vocabulary ob- ject detection using early dense alignment","venue":null,"work_id":"c8b8cbcd-ea6a-4dd7-b001-e78e941ea944","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.141043Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:fe4f003d5998180e3635966ee1e39cfd23443487ddfa37b61e7ce95f4465f9aa","observation_id":"64bff08c-1939-4e36-82e3-246c1dc24698","resolution":{"observed_at":"2026-08-12T18:02:15.664087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.641837Z","title":"Reco: Retrieve and co-segment for zero-shot transfer","venue":null,"work_id":"f50db21a-6f74-40a3-9b5e-01fa9d3c2ba9","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.145523Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:9bcfcb6d0091ab84e8b924ceb9279ba15dcffaca8e483e5b03425913fa0f6bc2","observation_id":"f1331559-3d22-4fff-8da3-1fab9c5ad7a4","resolution":{"observed_at":"2026-08-12T18:02:15.647733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.627387Z","title":"Going denser with open-vocabulary part segmentation","venue":null,"work_id":"aeb8b0a2-c523-46a5-b272-fcdbd2e1d198","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.149822Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:5c9206cfc6747ad1786d698bf6c84584058a7798f2d5306d05be4b55870c4df9","observation_id":"275e2ec4-c1d3-415d-9948-5d5671d3821c","resolution":{"observed_at":"2026-08-12T18:02:15.632789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.609346Z","title":"Clip as rnn: Segment countless visual concepts without training endeavor","venue":null,"work_id":"754d740b-e985-48cb-9bc9-a2aee9daf8be","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.153522Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:67f8b56bf57b29900ab2ab9dbfe7bf652218dd23e9e9774eb9aa048c20d69941","observation_id":"b5a22c4c-0561-46b1-80bb-9ae040ee9baf","resolution":{"observed_at":"2026-08-12T18:02:15.614395Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.580772Z","title":"Galip: Generative adversarial clips for text-to-image synthe- sis","venue":null,"work_id":"4b8e85f9-97f6-4954-95e4-65da42374f00","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.157867Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:051b2dcb6fca5a1798bd00e5d59d90b3a57a534b10164c34ef56a58e10bcaee1","observation_id":"bcc374ff-4d67-4c54-96a5-7b86198e0e83","resolution":{"observed_at":"2026-08-12T18:02:15.587655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.162625Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.162625Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:c55310b133998f9214b082b6ed3f3b4681d06af1939c85d185f9bedd4bbf86a6","observation_id":"f8581781-ecde-49f0-966c-776d4b62e2f6","resolution":{"observed_at":"2026-08-12T18:02:15.162625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01597","last_updated":"2024-10-26T15:58:10Z","snapshot_observed_at":"2026-08-16T14:38:00.694387Z","submitted_at":"2023-12-04T03:18:46Z","title":"SCLIP: Rethinking Self-Attention for Dense Vision-Language Inference","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.01597","snapshot_observed_at":"2026-08-12T18:02:15.166483Z","title":"Sclip: Rethink- ing self-attention for dense vision-language inference","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.166483Z"},"links":{"cited_paper":"/paper/2312.01597","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:22fc723278ec981bde16e928395f6432c185e12505fa2d81e8077657be7280e4","observation_id":"d9606a73-0f19-4eea-9f67-e31315b3625e","resolution":{"observed_at":"2026-08-12T18:02:15.166483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.00386","last_updated":"2022-03-01T12:11:32Z","snapshot_observed_at":"2026-08-16T17:17:50.020394Z","submitted_at":"2022-03-01T12:11:32Z","title":"CLIP-GEN: Language-Free Training of a Text-to-Image Generator with CLIP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.00386","snapshot_observed_at":"2026-08-12T18:02:15.171099Z","title":"Clip-gen: Language-free training of a text-to-image genera- tor with clip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.171099Z"},"links":{"cited_paper":"/paper/2203.00386","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:d41e3976f9e7021f2281c90ff19be0a7f40d2bd6f99818b0b144904e904c86ff","observation_id":"feb0ec46-cbde-4eec-955d-825dc8e0ba31","resolution":{"observed_at":"2026-08-12T18:02:15.171099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.549837Z","title":"Clip-diy: Clip dense infer- ence yields open-vocabulary semantic segmentation for-free","venue":null,"work_id":"34ed152a-a8b7-4e88-a736-d9dcbfae74d6","year":2024},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.179084Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:b41d295e67f08a908624180d512420973375e8d49d4b3fc31053901748934f70","observation_id":"02361cb0-5e06-4ee1-ad18-e1e7941c7d63","resolution":{"observed_at":"2026-08-12T18:02:15.555703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16671","last_updated":"2025-11-23T00:34:43Z","snapshot_observed_at":"2026-08-17T12:17:37.741844Z","submitted_at":"2023-09-28T17:59:56Z","title":"Demystifying CLIP Data","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16671","snapshot_observed_at":"2026-08-12T18:02:15.183230Z","title":"Demystify- ing clip data","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.183230Z"},"links":{"cited_paper":"/paper/2309.16671","citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:5e2c9620ce82e3f506b48de09fdd5cc7730c9b0b96726ec5538eecfb3f304445","observation_id":"a58c2522-69dc-41e7-a956-422f5e3d408a","resolution":{"observed_at":"2026-08-12T18:02:15.183230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.535367Z","title":"Groupvit: Semantic segmentation emerges from text supervision","venue":null,"work_id":"755c610e-a265-423e-b88e-696c818c63ba","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.187351Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:a5d6775fe97ab97896fbf255a8e69be283064c38d68a7577734bc07fae200f5f","observation_id":"2b0f967f-b902-4c88-9c79-c0cd8f061570","resolution":{"observed_at":"2026-08-12T18:02:15.540316Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.521009Z","title":"Learning open-vocabulary semantic segmentation models from natural language supervision","venue":null,"work_id":"c3544cbd-9e84-4b56-99e6-a4a6e3c6f469","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.191041Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:6e6f2458ae9de7649b8c85d677a77e7db93067ce72d2aec333aeb5df9cc04230","observation_id":"308f03c9-f66e-4035-9c47-0035234cf647","resolution":{"observed_at":"2026-08-12T18:02:15.525991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.194983Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.194983Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:25f77dcdb053f9b7ffe35a823168ac9594dc258a942a89becdbf2be6a00bff02","observation_id":"3f61d6fc-b9ee-49aa-a0e4-e03d8475c2ab","resolution":{"observed_at":"2026-08-12T18:02:15.194983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.480822Z","title":"A simple baseline for open- vocabulary semantic segmentation with pre-trained vision- language model","venue":null,"work_id":"42384eb2-d43a-47a5-a434-7e754066b255","year":2022},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.200520Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:0a890ffdaa4485dd898eb0aac405ec5fecfc0e701310a79114d609b573b2bfe0","observation_id":"5a54b700-5e61-4465-ab40-2ce190c15b47","resolution":{"observed_at":"2026-08-12T18:02:15.492430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.465640Z","title":"Learning deep features for discrimi- native localization","venue":null,"work_id":"922f350c-8774-4947-ba0b-d6f4e047f13c","year":2016},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.207065Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:3c669a7e07d1c846210fc73124b3e564c2a9bf79ece971689970128ce4a4cead","observation_id":"64285afd-c765-4523-b4fd-10a1301617e5","resolution":{"observed_at":"2026-08-12T18:02:15.471607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.212841Z","title":"Extract free dense labels from clip","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.212841Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:7a4aec9bb723cda03e8e594f051d66dcc98d6cd8770265176b5820facf68abd6","observation_id":"f0908021-ee5c-4ca8-b8e1-68d43e7cfc9b","resolution":{"observed_at":"2026-08-12T18:02:15.212841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:02:15.436728Z","title":"thing” classes and one explicit background class, our method fails to distinguish foreground classes from the background when the word “background","venue":null,"work_id":"fa2de428-2cf5-404b-8458-3b2fb6e77732","year":2023},"citing_paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T18:02:15.217453Z"},"links":{"citing_paper":"/paper/2411.12044"},"observation_digest":"sha256:4c2db8351501154cd73ca99404bcf55efcd3ed1ca1154d65e6085e3ba40c786b","observation_id":"6d1bb6a9-5310-4e4f-95b2-6d6dfa814bcd","resolution":{"observed_at":"2026-08-12T18:02:15.445785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.12044","last_updated":"2025-04-14T16:02:15Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T12:18:23.191452Z","submitted_at":"2024-11-18T20:31:38Z","title":"ITACLIP: Boosting Training-Free Semantic Segmentation with Image, Text, and Architectural Enhancements"},"reference_resolution":{"displayed":71,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":0,"verified_fuzzy":41},"total_outbound_references":71},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 71 of 71 outbound references and 2 inbound Pith citation observations for arXiv:2411.12044."}