{"as_of":"2026-08-13T03:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:55cb21098cec8aa0107ea32d7b09797b5755c3834e0c061d5c8ec6b557a63654","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T15:53:31.889301Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T20:04:44.106636Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-13T22:21:15.797283Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13836","snapshot_observed_at":"2026-08-12T20:04:44.106636Z","title":"Cliper: Hierarchically improving spatial represen- tation of clip for open-vocabulary semantic segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10086","last_updated":"2025-08-01T08:25:34Z","snapshot_observed_at":"2026-08-12T19:56:56.710365Z","submitted_at":"2024-11-15T10:14:55Z","title":"CorrCLIP: Reconstructing Patch Correlations in CLIP for Open-Vocabulary Semantic Segmentation","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T20:04:44.106636Z"},"links":{"cited_paper":"/paper/2411.13836","citing_paper":"/paper/2411.10086"},"observation_digest":"sha256:2681dea9191ac7a853a438f4d5aa71c68a997bde7121ebb35026ecacd33dbd7d","observation_id":"a7d5046b-d83a-4772-9b7f-346ca2037fe4","resolution":{"observed_at":"2026-08-12T20:04:44.106636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"cited_work":{"arxiv_id":"2411.13836","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13836","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"CLIPer: Hierarchically improving spatial representation of CLIP for open-vocabulary semantic segmentation","venue":null,"work_id":"51b1928f-fa27-4f57-8134-9f4b0d2e5e33","year":2024},"citing_paper":{"arxiv_id":"2504.13181","last_updated":"2025-04-28T18:01:39Z","snapshot_observed_at":"2026-08-11T19:32:15.215968Z","submitted_at":"2025-04-17T17:59:57Z","title":"Perception Encoder: The best visual embeddings are not at the output of the network","version":2},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-05-13T22:21:15.681336Z"},"links":{"cited_paper":"/paper/2411.13836","citing_paper":"/paper/2504.13181"},"observation_digest":"sha256:d95ad605cdb5b6891d1c4cf202ffa8b162a917c9b73a0fae41cd83837642d5cc","observation_id":"1fee426a-3576-4ef6-b5f8-473170d116dc","resolution":{"observed_at":"2026-05-13T22:21:15.798787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.13836/citation-record","integrity":"/paper/2411.13836/integrity","json":"/paper/2411.13836/citation-record.json","paper":"/paper/2411.13836"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:35.071758Z","title":"Weakly su- pervised learning of instance segmentation with inter-pixel relations","venue":null,"work_id":"5ec0f6b6-a556-408c-b6f0-9f3fe274eb0b","year":2019},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.953476Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:8c4425fbbca204807942c3bc5fa950554c2ffa6ceff8777d59fd2943924c5841","observation_id":"e8d4222c-c7a9-42dd-8f17-90c0bb26a74d","resolution":{"observed_at":"2026-08-12T15:53:35.104096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.967475Z","title":"Zero-shot semantic segmentation","venue":null,"work_id":"f3dc0727-f386-4ca5-a37d-a11c59d07569","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.960642Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:22f467159c4601207ebec976d4818526f076734fb194ff15f55010a7f68e4dc1","observation_id":"6c6c5a6a-84a4-4045-aa37-4967a6e6d61d","resolution":{"observed_at":"2026-08-12T15:53:35.009572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.878577Z","title":"Coco- stuff: Thing and stuff classes in context","venue":null,"work_id":"ea67ebe2-0c84-4293-92dd-29e12dd7c464","year":2018},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.967487Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:32526fed6f259417da3c6d52f31526cc3c87cb8d527d2b0b66c8de69efb3af52","observation_id":"d18a48e3-a586-4cd5-b360-d17f0e54e0a9","resolution":{"observed_at":"2026-08-12T15:53:34.916255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.828422Z","title":"Emerg- ing properties in self-supervised vision transformers","venue":null,"work_id":"9698c3b4-1acb-4047-ab46-b92982c838ff","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.973143Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:121dae4eef88f7ff0433931dc3171df4aabca9797f05b38339ba55d9bfcccda5","observation_id":"197c4d1a-34e0-4973-abff-865a2fdd9132","resolution":{"observed_at":"2026-08-12T15:53:34.852705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.768051Z","title":"Learn- ing to generate text-grounded mask for open-world semantic segmentation from only image-text pairs","venue":null,"work_id":"4b36e1f9-e0c7-4f6a-9126-2990de5927e9","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.980180Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:9d440b14d75e652803433f6a5acdf129f835a1cceec707d238de2152617cf0db","observation_id":"2fb29648-8d04-4c9b-b931-1e7a56e978bd","resolution":{"observed_at":"2026-08-12T15:53:34.806179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.732584Z","title":"Masked-attention mask transformer for universal image segmentation","venue":null,"work_id":"8a04b96b-b6b0-4762-915d-1c8e435ed6e6","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:29.992618Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:6b9c66730bd95363309c204c3d4381bff1292e29684a697d88d81ef9ef5b226c","observation_id":"dfd3e05b-4488-4f30-9402-a96375d05028","resolution":{"observed_at":"2026-08-12T15:53:34.742847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.691891Z","title":"Reproducible scaling laws for contrastive language-image learning","venue":null,"work_id":"0e836120-e267-4ddd-a73d-f2205b759c13","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.005895Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:10f6b6ea6582a6f0d982bc8f78201d4ce24d25c626ffdb57ee65802a95c72374","observation_id":"bfa636b9-7eb7-4292-983e-64c112e2cc83","resolution":{"observed_at":"2026-08-12T15:53:34.704693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11797","last_updated":"2024-03-31T11:53:55Z","snapshot_observed_at":"2026-08-10T15:40:12.549701Z","submitted_at":"2023-03-21T12:28:21Z","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11797","snapshot_observed_at":"2026-08-12T15:53:30.079121Z","title":"Cat-seg: Cost aggregation for open-vocabulary semantic segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.079121Z"},"links":{"cited_paper":"/paper/2303.11797","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:ea085537ac6a295fcb312ea299141349a3adf386a6ff4b1ce1b5ff1bb8c92b97","observation_id":"5272cc8f-b55b-45cd-92b1-ba3b62e3fe37","resolution":{"observed_at":"2026-08-12T15:53:30.079121Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2208.08984","last_updated":"2023-06-08T06:35:33Z","snapshot_observed_at":"2026-07-06T13:43:13.386359Z","submitted_at":"2022-08-18T17:55:37Z","title":"Open-Vocabulary Universal Image Segmentation with MaskCLIP","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.08984","snapshot_observed_at":"2026-08-12T15:53:30.129614Z","title":"Open- vocabulary panoptic segmentation with maskclip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.129614Z"},"links":{"cited_paper":"/paper/2208.08984","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:bc9251903fafc587fb5bfecef89b553e584bca1d29c1a73b2aa06782583f1b9a","observation_id":"5c3e7584-fa90-43a4-8185-0a4fe387b6ed","resolution":{"observed_at":"2026-08-12T15:53:30.129614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T02:40:23.887636Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T15:53:30.187878Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.187878Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:c9b2df64225b233115e993357a20622795a06df7bb069e37d20276264020db49","observation_id":"d277024b-19b1-4fc2-b1f8-077edea43f61","resolution":{"observed_at":"2026-08-12T15:53:30.187878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.660755Z","title":"The pascal visual object classes (voc) challenge","venue":null,"work_id":"fbdeedbf-2d56-4b12-9a24-fc6df496e3b1","year":2010},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.241754Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:0ee87a032d7b332ff678ffe361f5a83f29448122f757eafc2a7abdc976b3c840","observation_id":"78f240fe-ec8b-4796-aadf-3b4b65160abe","resolution":{"observed_at":"2026-08-12T15:53:34.669246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.619532Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":"00de7dda-f664-43ea-a21f-608878938c1e","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.253622Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:ec4c43a39c4d861f9945703ee45394a1acd9cfe6c46ac0997da90b419820e61d","observation_id":"0881ea7d-4ff2-4390-8134-9d8081342a30","resolution":{"observed_at":"2026-08-12T15:53:34.636723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.589301Z","title":"Diffusion models for zero-shot open-vocabulary segmentation","venue":null,"work_id":"c1381a04-b7ae-4b95-9d64-1e12a3778632","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.262474Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:8026a7f9ffae8262121853b02a033065c61de2b7b6c70c8c7b649ad66e29226a","observation_id":"2274c32d-a5ad-4847-8272-79a291cfca52","resolution":{"observed_at":"2026-08-12T15:53:34.594676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.552160Z","title":"Segment anything","venue":null,"work_id":"1b2f621b-2b91-48d1-8a17-f94a01db2b85","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.323897Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:9aa24f379ce6b75e9338c34a6dc1c8391faf3612af27b3db8698a416251374a6","observation_id":"6b827282-485e-49ef-88eb-515e84b63026","resolution":{"observed_at":"2026-08-12T15:53:34.561979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.511683Z","title":"ProxyCLIP: Proxy attention improves clip for open-vocabulary segmentation","venue":null,"work_id":"23354389-5a0e-4ec9-a217-32ba836ef273","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.434077Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d6034b4af9499b25b97834601d0f39d8d42739c15de360293a4088af98bbbc29","observation_id":"423ec143-ef68-46a0-8dd1-e6d917960360","resolution":{"observed_at":"2026-08-12T15:53:34.527327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.393175Z","title":"ClearCLIP: Decom- posing clip representations for dense vision-language infer- ence","venue":null,"work_id":"81a32e6a-df33-4468-83de-8fa7f72f0d47","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.511695Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:621b6b11e48aae20d2f2756e05a283264e6995a6fdaee904e28de3eeef898730","observation_id":"6dc9b584-b0b6-4919-87f8-cb622f46ef15","resolution":{"observed_at":"2026-08-12T15:53:34.469805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.340745Z","title":"Anti- adversarially manipulated attributions for weakly and semi- supervised semantic segmentation","venue":null,"work_id":"7e355020-f664-4cb0-99d8-d1144f30761b","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.524335Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:0532f1b736cab2a6d8055c934292717eda8aa230eb784e4d58d97e7ca31704e0","observation_id":"0e733d73-93f9-4fc4-b743-8420134f2b27","resolution":{"observed_at":"2026-08-12T15:53:34.352659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05653","last_updated":"2024-09-16T09:10:00Z","snapshot_observed_at":"2026-08-12T06:50:50.066504Z","submitted_at":"2023-04-12T07:16:55Z","title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05653","snapshot_observed_at":"2026-08-12T15:53:30.578203Z","title":"A closer look at the explainability of contrastive language-image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.578203Z"},"links":{"cited_paper":"/paper/2304.05653","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:bdfc13bb68dbfcd06282c25703f94375e771162744bcffe231b0f27cfb3776e2","observation_id":"9fa5cd2c-22c7-4835-9522-776468f09def","resolution":{"observed_at":"2026-08-12T15:53:30.578203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.294257Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":"0258942c-3945-4216-93e8-ffba48fd9683","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.672403Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:427ac14ee31bc40a6a275cee61cca8c1f213e0133fa97428c173f135a4e5fafc","observation_id":"29a748cd-ff89-4e14-8ada-6c5fa9bea170","resolution":{"observed_at":"2026-08-12T15:53:34.308320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.262440Z","title":"CLIP is also an efficient segmenter: A text-driven approach for weakly su- pervised semantic segmentation","venue":null,"work_id":"45c99e5d-d3cc-43e0-be57-938ea56bbd3c","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.744405Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:a1c5de8b36ce65fe38225902eaf57eceb6cdb2b9c148c89eb9bfc03546937605","observation_id":"cafa38ba-9d28-4b8a-a449-f375d9a8c235","resolution":{"observed_at":"2026-08-12T15:53:34.274869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.227983Z","title":"TagCLIP: A local-to-global framework to enhance open-vocabulary multi-label classification of clip without training","venue":null,"work_id":"cdfde013-3cf6-4ac1-b38c-c6d32e9c8ef5","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.756293Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:b2b31a448cf054181e639bfcc9a31e87f0b539d4039f980d230ac3960ec38838","observation_id":"449e98ac-1963-48a5-babc-1bd3c2baeffb","resolution":{"observed_at":"2026-08-12T15:53:34.237480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.194916Z","title":"Fully convolutional networks for semantic segmentation","venue":null,"work_id":"3829924a-c49c-410f-ab57-66614bdfdbd6","year":2015},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.764526Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d1c24c72d1c2d817cc478ae635d2ed4e5cdf4a8ebb07a7e38760dad9080352cc","observation_id":"9744087a-2167-4d00-b9a8-cbc741bbe050","resolution":{"observed_at":"2026-08-12T15:53:34.201120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.157370Z","title":"SegCLIP: Patch aggregation with learnable centers for open-vocabulary semantic segmentation","venue":null,"work_id":"09021794-b224-4362-bd5a-8f188af55c9b","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.854185Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:b384151c42118a833c8343f517a3534665e170f8ae51f74dc92203916e9215ec","observation_id":"66449b3c-c91c-4a61-8ca8-18a167b6a519","resolution":{"observed_at":"2026-08-12T15:53:34.169300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:34.113034Z","title":"The role of context for object detection and semantic segmentation in the wild","venue":null,"work_id":"81b8cb23-2e23-43b3-b689-e85ced8155bb","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.956732Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:3c342c6c83fe261f0df43238ffe77013961299c1d83faafcc48689a54dc72c8c","observation_id":"4cbf94d9-3b2a-4eb4-9e34-5ee0d946aa24","resolution":{"observed_at":"2026-08-12T15:53:34.127614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.901696Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"1e5bf99f-3473-4987-a9a6-fb7ffb2c0d49","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:30.980568Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:3439f9d249c25163abe126e00449da45722b9e11259fb8ad5859098c9679a901","observation_id":"82a288b3-01c7-41c6-9ee1-579ecf5ed70b","resolution":{"observed_at":"2026-08-12T15:53:33.983972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.851467Z","title":"Per- ceptual grouping in contrastive vision-language models","venue":null,"work_id":"2f8f1d38-6578-43b9-8eb4-9341f8d99f27","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.004171Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:5e4403eb3d278917d1c7d95751ace89bc2ab488a5d19512dc625273dd0961717","observation_id":"d3175b13-b7f6-4075-a543-9417b6f3b717","resolution":{"observed_at":"2026-08-12T15:53:33.861006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.807847Z","title":"ViewCo: Discovering text-supervised segmentation masks via multi-view semantic consistency","venue":null,"work_id":"c6cd168e-6473-4eae-a4dc-d78e3281375a","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.016261Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d5eed8229c874b67d4746649e76a2546e32c72251afbcfdd33f8071c2b9e81e3","observation_id":"e7ea2c76-8fe9-4204-97ac-b02c22479857","resolution":{"observed_at":"2026-08-12T15:53:33.818391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.773559Z","title":"High-resolution image syn- thesis with latent diffusion models","venue":null,"work_id":"ee6deaff-52c1-49b5-81e4-df93aff0fc0c","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.091640Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:3d22ca5c6d4b36055934a43a893a94c11ad8a2560b0254f68b28d3c2ef0b9491","observation_id":"dc4bb987-e0bd-48da-8858-9c13f054bca6","resolution":{"observed_at":"2026-08-12T15:53:33.785884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.738026Z","title":"To- ken contrast for weakly-supervised semantic segmentation","venue":null,"work_id":"df8e0a80-ca45-4558-ba35-fe12adc86641","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.205206Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:df76c3982a4e4c3f31ddcc63d5a9e3291a7fa286e5be5d2eeb971f5ae298d8ab","observation_id":"8db690b3-cad4-42db-9946-f1bf9512de3a","resolution":{"observed_at":"2026-08-12T15:53:33.749028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.702716Z","title":"Laion-5B: An open large-scale dataset for training next generation image-text models","venue":null,"work_id":"faf43ba7-d316-44af-8190-1740da79bb5d","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.270985Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:a74b7d5e2cfc5c8ea9ede6783d0dad64ceab21c2433dc4efd8c58e6b3f0fd7a1","observation_id":"ff6fa4f0-1142-43e1-ac19-a7faab042712","resolution":{"observed_at":"2026-08-12T15:53:33.713015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.662385Z","title":"ReCo: Re- trieve and co-segment for zero-shot transfer","venue":null,"work_id":"b4c5bc70-8983-455a-80a4-45323cf79ada","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.288916Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:57134e62e6c2cbcd090c2a7bbc9fc88a1890cf7523611f07d82c0aaab5f5925f","observation_id":"3ef15bd6-675a-46e3-ae5b-d66e8b41b6db","resolution":{"observed_at":"2026-08-12T15:53:33.672270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03209","last_updated":"2024-10-08T08:32:36Z","snapshot_observed_at":"2026-08-12T22:51:04.355719Z","submitted_at":"2024-09-05T03:07:26Z","title":"iSeg: An Iterative Refinement-based Framework for Training-free Segmentation","version":4},"cited_work":{"arxiv_id":"2409.03209","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.03209","snapshot_observed_at":"2026-08-12T15:53:32.117301Z","title":"iSeg: An Iterative Refinement-based Framework for Training-free Segmentation","venue":"cs.CV","work_id":"7ce38b09-3ecc-408e-a74e-e078db4a6b0c","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.299753Z"},"links":{"cited_paper":"/paper/2409.03209","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:078b2522517ddc562ad51526a45eb8c95a37b4b9d0315f812f4d8511387bdf8f","observation_id":"76af3956-dcb7-43e3-82b4-dade04414c75","resolution":{"observed_at":"2026-08-12T15:53:32.128129Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.624187Z","title":"CLIP as RNN: Segment countless visual concepts with- out training endeavor","venue":null,"work_id":"a1026133-8569-40cb-9ab1-c2e9613b3bc1","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.324636Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:484616661bdc1ece8000e6d1203b2206eb2064d6ee95bfd46eccc1ebedbfd458","observation_id":"ee87535e-4d4b-41ad-8387-f8c20bdb7c73","resolution":{"observed_at":"2026-08-12T15:53:33.634456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.462816Z","title":"Sclip: Rethinking self-attention for dense vision-language inference","venue":null,"work_id":"cbe222b4-971f-4a1b-a813-68f41181ef87","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.421744Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:b2ba1a27b866f2fa36a34cb38cd3b129cb46bdfb8bc1e16c8943a9b879f89079","observation_id":"0e62b1ef-f7ee-410a-9bee-51decc852a1f","resolution":{"observed_at":"2026-08-12T15:53:33.574412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.386515Z","title":"Sam-clip: Merging vision foundation models towards semantic and spatial understanding","venue":null,"work_id":"3df78d9e-8e00-4d0f-aa5d-3a3804f651c3","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.466108Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:f05d95891fbed5de85f67a5e076835afaa7543d6ce5fa55cb69a23b13159bc0f","observation_id":"537c9fe7-9180-47fc-beb5-44f643700f30","resolution":{"observed_at":"2026-08-12T15:53:33.398863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.02773","last_updated":"2024-01-22T07:18:55Z","snapshot_observed_at":"2026-08-07T23:37:35.393306Z","submitted_at":"2023-09-06T06:31:08Z","title":"Diffusion Model is Secretly a Training-free Open Vocabulary Semantic Segmenter","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.02773","snapshot_observed_at":"2026-08-12T15:53:31.546345Z","title":"Diffusion model is secretly a training-free open vocabulary semantic segmenter","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.546345Z"},"links":{"cited_paper":"/paper/2309.02773","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:fb763bd8bb96e87998c835f4bc5d60b70e6553965d218a78aef5154ce2c270ca","observation_id":"f1e187ce-2dfd-4150-8125-3768a669df13","resolution":{"observed_at":"2026-08-12T15:53:31.546345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.232759Z","title":"Clip-dinoiser: Teaching clip a few dino tricks for open- vocabulary semantic segmentation","venue":null,"work_id":"3c002ebf-4a5a-4187-a87c-495ebc334456","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.678102Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:447b3a617d9d0005740bab565d6c71714c4dd514a722117d46717f6ce637eebf","observation_id":"e021f2c3-832d-4647-886f-e9733f0fac57","resolution":{"observed_at":"2026-08-12T15:53:33.361545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.04109","last_updated":"2024-10-01T10:30:07Z","snapshot_observed_at":"2026-07-06T16:15:58.653512Z","submitted_at":"2023-09-08T04:10:01Z","title":"From Text to Mask: Localizing Entities Using the Attention of Text-to-Image Diffusion Models","version":2},"cited_work":{"arxiv_id":"2309.04109","doi":null,"metadata_source":"pith","pith_arxiv_id":"2309.04109","snapshot_observed_at":"2026-08-12T15:53:32.028523Z","title":"From Text to Mask: Localizing Entities Using the Attention of Text-to-Image Diffusion Models","venue":"cs.CV","work_id":"6db47440-5bd3-4a74-a2af-14863c4d90c0","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.718299Z"},"links":{"cited_paper":"/paper/2309.04109","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:3320e601209925bd66a3a3fdf435df5989288510aafaa655e31e667dbb32a467","observation_id":"5268430a-a04e-4686-bde8-0acc42f79d99","resolution":{"observed_at":"2026-08-12T15:53:32.041705Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.194200Z","title":"Sed: A simple encoder-decoder for open- vocabulary semantic segmentation","venue":null,"work_id":"8d4f1c10-f1db-4a1d-9b0a-c545b5abf82a","year":2024},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.738935Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:4753b951510ad4b47c7f519991ea5bf1c6a7893a7b17c65ff5e573de2cbab36e","observation_id":"0fd812d0-79ab-49e5-8126-9f64443fd613","resolution":{"observed_at":"2026-08-12T15:53:33.203098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:33.088063Z","title":"Alvarez, and Ping Luo","venue":null,"work_id":"c9abf15e-04d4-41e8-9710-31acae16f91a","year":2021},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.752197Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:8e2b03b30af37944b63b9f1442d85abc040ce5e29ba058aecf7016572ea20d64","observation_id":"6818b0d7-7fbb-4083-b73b-4807879449f2","resolution":{"observed_at":"2026-08-12T15:53:33.169597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.940660Z","title":"Clims: Cross language image matching for weakly supervised se- mantic segmentation","venue":null,"work_id":"aee72a42-05d7-4945-b6f3-bf2bc1c4ebc5","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.765289Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:3712abc53ca7cc057856ffea4bcf150bf5f64c75985ecc742a26e85e191edeb0","observation_id":"82e486ea-48b1-475c-bb02-2c9994ba1305","resolution":{"observed_at":"2026-08-12T15:53:32.990468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.892729Z","title":"Rewrite caption semantics: Bridging seman- tic gaps for language-supervised semantic segmentation","venue":null,"work_id":"c103edb7-5e7a-44fe-95ce-474df54aeef8","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.782187Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d5dc419b848bf41421d3cc6accf26f38d5ca3ed01570517c1b9705db2e3bb0b7","observation_id":"179b678d-4304-41da-b23a-b38123aa2586","resolution":{"observed_at":"2026-08-12T15:53:32.915948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.841835Z","title":"Groupvit: Semantic segmentation emerges from text supervision","venue":null,"work_id":"26f961a9-ab64-4f4b-b686-8f7d6952bd31","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.798853Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d67baec0f95eb99472960dc8eff010132725229decc4b6e356ae1ffb5b2a0724","observation_id":"aa91ddfa-5545-4106-83cb-1dbe4aaf9647","resolution":{"observed_at":"2026-08-12T15:53:32.850323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.800365Z","title":"Learning open-vocabulary seman- tic segmentation models from natural language supervision","venue":null,"work_id":"75c7f5ed-9018-44c5-a027-12ef66c5513c","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.808098Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:5c7f0b308323e7bbb22ba0fa74acd66fb4e2fe2c6c3f49b035b0147e53c9f8eb","observation_id":"34810fe9-a51d-4920-83db-c970a2641733","resolution":{"observed_at":"2026-08-12T15:53:32.809151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.763433Z","title":"Open-vocabulary panoptic segmentation with text-to-image diffusion models","venue":null,"work_id":"f5a63cd0-ac4c-4341-be83-f2f8644fc4d7","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.817712Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:6a58ec6541b373701399d9b4bcb098b7339dee712b2bb42ae499b5a568a396d7","observation_id":"2f00c189-f6a3-4241-a794-bda1d061773d","resolution":{"observed_at":"2026-08-12T15:53:32.773430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.693071Z","title":"Multi-class token transformer for weakly supervised se- mantic segmentation","venue":null,"work_id":"2e2e7efe-df80-4b01-94ba-d4b13b09352c","year":null},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.824494Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:0cdbb2c75300f941c3a252eb5cbd878f897725b13ea962f2488462aca30be24b","observation_id":"16223171-313d-4031-a611-a09d431217d6","resolution":{"observed_at":"2026-08-12T15:53:32.726561Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.647886Z","title":"Side adapter network for open-vocabulary semantic segmentation","venue":null,"work_id":"69885f3f-bfcc-414e-aac8-63b47a6c6626","year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.838226Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:d5b9318d2681fe31f0799b195dc829447a8a1ea3f5a7f2cb8f76677acaff0f64","observation_id":"0732b187-27e6-4bd9-9ecd-79a5528df144","resolution":{"observed_at":"2026-08-12T15:53:32.660574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02487","last_updated":"2023-11-14T19:10:49Z","snapshot_observed_at":"2026-08-04T01:52:38.789872Z","submitted_at":"2023-08-04T17:59:01Z","title":"Convolutions Die Hard: Open-Vocabulary Segmentation with Single Frozen Convolutional CLIP","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.02487","snapshot_observed_at":"2026-08-12T15:53:31.853023Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.853023Z"},"links":{"cited_paper":"/paper/2308.02487","citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:2b0d5b2b5c689310d927582c09ca7d2c1019fa292e48042ffff5a130436b69ec","observation_id":"dedbb98a-65d2-424e-b811-32470ae6b714","resolution":{"observed_at":"2026-08-12T15:53:31.853023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.588332Z","title":"Open vocabulary scene parsing","venue":null,"work_id":"f8a3eb35-17d5-4c6c-a233-193beff8b4f2","year":2002},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.865201Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:fcc60cce963bd4dbd91ea6d1b55859b44e6c7e5fcc003b34240445c827ee7641","observation_id":"91ee6cf4-fc84-47f8-bc5f-272acd54846b","resolution":{"observed_at":"2026-08-12T15:53:32.616043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:31.878658Z","title":"Semantic under- standing of scenes through the ade20k dataset","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.878658Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:525afd5d6c995e14947cf71fec0c1a23f6299fc4cb08cd1d34651a05769ee24e","observation_id":"c5655f20-8b1e-4db0-8adb-12423fbf6252","resolution":{"observed_at":"2026-08-12T15:53:31.878658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:32.401277Z","title":"Extract free dense labels from clip","venue":null,"work_id":"dc71af18-c840-4af3-bc7e-87385c42e1b3","year":2022},"citing_paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:31.889301Z"},"links":{"citing_paper":"/paper/2411.13836"},"observation_digest":"sha256:1c4ade0bee2bb0b884feed17bee2249cc05817218ed63526cd8831e952efe7f9","observation_id":"3241dc5e-106a-4fec-82df-4def85422460","resolution":{"observed_at":"2026-08-12T15:53:32.461065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.13836","last_updated":"2024-11-21T04:54:30Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T15:47:03.465474Z","submitted_at":"2024-11-21T04:54:30Z","title":"CLIPer: Hierarchically Improving Spatial Representation of CLIP for Open-Vocabulary Semantic Segmentation"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":2,"verified_fuzzy":42},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 2 inbound Pith citation observations for arXiv:2411.13836."}