{"as_of":"2026-08-07T18:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:49c2b0d4c3b045583e234078a51f1061adbb59c06180639c4e239f4a4f4550c0","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T23:50:23.565560Z","state":"measured"},{"denominator":66,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":66,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.16058/citation-record","integrity":"/paper/2506.16058/integrity","json":"/paper/2506.16058/citation-record.json","paper":"/paper/2506.16058"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.250219Z","title":"Self-calibrated clip for training-free open-vocabulary segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.250219Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ff8dac0b7ed2b5f8d47e4052ea8211b899aa29fa97d5aaa4199e13e90e19350c","observation_id":"a97559ff-9004-4296-ab05-ec06c700efd3","resolution":{"observed_at":"2026-08-06T23:50:23.250219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14231","last_updated":"2025-05-20T11:40:43Z","snapshot_observed_at":"2026-08-07T15:35:48.142895Z","submitted_at":"2025-05-20T11:40:43Z","title":"UniVG-R1: Reasoning Guided Universal Visual Grounding with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14231","snapshot_observed_at":"2026-08-06T23:50:23.255593Z","title":"Univg-r1: Reasoning guided universal visual grounding with reinforce- ment learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.255593Z"},"links":{"cited_paper":"/paper/2505.14231","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:9b71c33256ee0017d1c3749f883b8f7a3a6ca3ac6616bc3cbe3c350d54f70300","observation_id":"8ddce7f4-5588-41a2-83f5-52828cd4f189","resolution":{"observed_at":"2026-08-06T23:50:23.255593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.721296Z","title":"Brostow, Jamie Shotton, Julien Fauqueur, and Roberto Cipolla","venue":null,"work_id":"3c602268-fabe-401c-a0b7-c484e22fb9ba","year":2008},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.261190Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:97956aeda5f22942f0e4cef495ea9e083e7c36d5a2fd51f91aa3569af59ab2bf","observation_id":"97e86502-505d-4e63-a351-faf64d93292e","resolution":{"observed_at":"2026-08-06T23:50:24.726791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.706160Z","title":"Zero-shot semantic segmentation","venue":null,"work_id":"d4ecdf32-e5a4-4e69-9db8-9624b33576b4","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.266818Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:50021f4dbeccc0386ab1009319a30527adf7da372d7e1cbc21ad19a411de0122","observation_id":"668c0a39-61f2-4d63-b2c7-837a4f063b19","resolution":{"observed_at":"2026-08-06T23:50:24.711443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.691013Z","title":"End-to- end object detection with transformers","venue":null,"work_id":"eae58f91-8ea4-4176-809e-b0b456964c6e","year":2020},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.271584Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:d22eae7b02c4b5f06fc8a7ea1647913fb5cff80d5acd20d23b47585a7e82bf2f","observation_id":"c50bfe99-7026-4e63-8648-f42be4d18a4a","resolution":{"observed_at":"2026-08-06T23:50:24.696514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.676901Z","title":null,"venue":null,"work_id":"51a0d026-7010-42bb-8cc6-80e11693b1af","year":2018},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.276968Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:2b17894b4ed9f39ea6ca821eef125ae9357a769bee214284f71792202aadba00","observation_id":"e38a447f-63b6-4c93-95d4-88c76ddd1afe","resolution":{"observed_at":"2026-08-06T23:50:24.681255Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11740","last_updated":"2020-07-17T22:19:59Z","snapshot_observed_at":"2026-08-07T11:11:46.670423Z","submitted_at":"2019-09-25T20:02:54Z","title":"UNITER: UNiversal Image-TExt Representation Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11740","snapshot_observed_at":"2026-08-06T23:50:23.281943Z","title":"UNITER: learning universal image-text representations","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.281943Z"},"links":{"cited_paper":"/paper/1909.11740","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:1bb12fb6a997a0d1964a0356b450e32f11eb446fe97c3e47ad94362a74e300f4","observation_id":"97cb7be6-a06c-463a-8359-e2eb0820f02c","resolution":{"observed_at":"2026-08-06T23:50:23.281943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.10764","last_updated":"2021-12-20T18:59:59Z","snapshot_observed_at":"2026-08-06T00:03:37.713050Z","submitted_at":"2021-12-20T18:59:59Z","title":"Mask2Former for Video Instance Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.10764","snapshot_observed_at":"2026-08-06T23:50:23.286782Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.286782Z"},"links":{"cited_paper":"/paper/2112.10764","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ec98596c4433d990290da6ba96b47297ba8a615ef8ae07745db50738c8c74bf5","observation_id":"cab951f1-9f45-4213-978b-f68eea78264e","resolution":{"observed_at":"2026-08-06T23:50:23.286782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.292230Z","title":"Schwing, Alexan- der Kirillov, and Rohit Girdhar","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.292230Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:8325c08e4971a9c55256dae49ef28e6407936b489e184f25fc997b9a3e6a8ae9","observation_id":"c5b765f5-6d38-48ca-89e7-f31e3eecfc23","resolution":{"observed_at":"2026-08-06T23:50:23.292230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.651884Z","title":"Cat-seg: Cost aggregation for open-vocabulary semantic segmenta- tion","venue":null,"work_id":"a88d2659-2b35-4136-8147-1b5fed565f48","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.297117Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:bbfe9e60cfcdadb572e27b26555be1213df6a92c1b908071d10cf91a58067465","observation_id":"d3f43add-dd73-459f-9379-39f466b93474","resolution":{"observed_at":"2026-08-06T23:50:24.656705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.301829Z","title":"The cityscapes dataset for semantic urban scene understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.301829Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:7ede5d97ba10219836c5d3ce3874c5da0ce7a3becf6e7b47b67c92b9a0e5001c","observation_id":"4723c5f7-8b26-4713-a00e-80203d198265","resolution":{"observed_at":"2026-08-06T23:50:23.301829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.626285Z","title":"De- coupling zero-shot semantic segmentation","venue":null,"work_id":"8dbd26db-ac5a-4fa3-952b-5c8c85274426","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.307089Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:c98dfdc71552b7b2c133a58fac004f1e893608b5f19092ab0ee7f06156f2c843","observation_id":"e4678ec7-af9e-4dda-ac7b-b7463e10667d","resolution":{"observed_at":"2026-08-06T23:50:24.630931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.610530Z","title":"The pascal visual object classes challenge: A retrospective.IJCV, 111:98–136, 2015","venue":null,"work_id":"d17ccd2f-0342-481b-8f4b-142b157c099a","year":2015},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.311413Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ec486c42234353962ad006955d45ca9d96f8794f578ce3bfb10248c3c21cd851","observation_id":"9f2bc313-d6c4-40a1-8738-f963b9a62445","resolution":{"observed_at":"2026-08-06T23:50:24.615216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.595394Z","title":"Large-scale unsu- pervised semantic segmentation","venue":null,"work_id":"be58d233-6201-4abc-9c46-b698b1cdc4f0","year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.316014Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:5cdc35cded180a46d68823dee59839cec039e2d815467fb8ab527ce33d456e3f","observation_id":"0d1d0ab4-f453-41f8-a227-bfef21cb9f24","resolution":{"observed_at":"2026-08-06T23:50:24.600200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.12143","last_updated":"2022-07-20T21:56:52Z","snapshot_observed_at":"2026-07-06T12:21:42.160571Z","submitted_at":"2021-12-22T18:57:54Z","title":"Scaling Open-Vocabulary Image Segmentation with Image-Level Labels","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.12143","snapshot_observed_at":"2026-08-06T23:50:23.320589Z","title":"Open-vocabulary image segmentation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.320589Z"},"links":{"cited_paper":"/paper/2112.12143","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:4661d102b531f793d1d4ef3166793012f49adaaeae5589eb5779b430d3d19c21","observation_id":"edc5a565-09b0-4bc9-8784-dc05196a8f99","resolution":{"observed_at":"2026-08-06T23:50:23.320589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.578767Z","title":"Random walks for image segmentation","venue":null,"work_id":"a6b56625-211c-43a8-a9cb-601a99e14347","year":2006},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.325420Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ba90e49412c87f3ac0a704ade6a055f7760935e24a8b4684572b85c9feea4854","observation_id":"9e39c27a-a7ff-4470-9077-2fd5cb6a8031","resolution":{"observed_at":"2026-08-06T23:50:24.584747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.562936Z","title":"Global knowledge calibration for fast open-vocabulary segmentation","venue":null,"work_id":"d432bfff-b32d-424c-b80e-d07d176da4d3","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.330380Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:7dd38e7b0c11533054eb0f0768a3cf559d811e7be31e8ac2c77b1b5604b5099d","observation_id":"39bb6abf-29bb-4c4f-8712-ea9fb4931231","resolution":{"observed_at":"2026-08-06T23:50:24.567916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.547262Z","title":"Primitive gener- ation and semantic-related alignment for universal zero-shot segmentation","venue":null,"work_id":"76c89121-5761-453e-8151-85147174b4a0","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.335212Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:63522cef9815d2fd033485f1ede0c3d2c1935d03c21a5a9ddb56ac172828767e","observation_id":"25db6e78-78b1-40f8-b904-20340d80d6c3","resolution":{"observed_at":"2026-08-06T23:50:24.552376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.340138Z","title":"Planning-oriented autonomous driving","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.340138Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b27d558fa09e5bda773a9ff090e1c8266b90b9cf1d29e8de1fc947493585d54a","observation_id":"b676ed18-8618-4d5d-a61e-2ef7499ba897","resolution":{"observed_at":"2026-08-06T23:50:23.340138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08580","last_updated":"2025-01-15T05:00:03Z","snapshot_observed_at":"2026-07-06T20:21:13.914417Z","submitted_at":"2025-01-15T05:00:03Z","title":"Densely Connected Parameter-Efficient Tuning for Referring Image Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08580","snapshot_observed_at":"2026-08-06T23:50:23.344365Z","title":"Densely connected parameter- efficient tuning for referring image segmentation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.344365Z"},"links":{"cited_paper":"/paper/2501.08580","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:536269d08101091fca1106216b12d5674f7f7c69b8e4438b9c745ae274a1552b","observation_id":"3268e0aa-7e79-441c-93b2-a1f7b2c8fd6c","resolution":{"observed_at":"2026-08-06T23:50:23.344365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00869","last_updated":"2024-03-26T12:56:55Z","snapshot_observed_at":"2026-07-06T16:55:49.813116Z","submitted_at":"2023-12-01T19:00:17Z","title":"Segment and Caption Anything","version":2},"cited_work":{"arxiv_id":"2312.00869","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.00869","snapshot_observed_at":"2026-08-06T23:50:23.790350Z","title":"Segment and Caption Anything","venue":"cs.CV","work_id":"2e7cc23b-8dc0-4956-9b7e-1b93f125232d","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.348944Z"},"links":{"cited_paper":"/paper/2312.00869","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:01b18e592ce2e78059f502c1b949332214686fca534a3a178dcbaf1ac144909b","observation_id":"de7f347b-2b56-4e45-98e4-5c4184e9f8af","resolution":{"observed_at":"2026-08-06T23:50:23.795324Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.520741Z","title":"Proxydet: Synthesizing proxy novel classes via classwise mixup for open-vocabulary object detection","venue":null,"work_id":"dd89ade7-4a95-4900-a548-0a3f1d06e255","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.354116Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:258d8fb06f9c225c89b929980002183e2a0b65d18f19087bd71706c49c067d0a","observation_id":"cc55fbdc-facc-4b62-86d5-2b7d9f022012","resolution":{"observed_at":"2026-08-06T23:50:24.525350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.506797Z","title":"Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig","venue":null,"work_id":"d9bafb52-259d-40ab-80df-54e27cf7838b","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.359615Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ff59ff6611d1e25fad1b10ac4c15c62504fc6da2161ae6fc223e22d246d52366","observation_id":"98fa837c-4a11-4694-8f06-5a56b65b9df6","resolution":{"observed_at":"2026-08-06T23:50:24.511326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.490349Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":"9b8e06e5-98ca-41ff-92f8-174d131aa7aa","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.364378Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:e52806ba9bc763c0bb16614d370d0fbf7f685dd7c553461e64823f91ab20b580","observation_id":"d147be5f-999c-4b9c-a8d1-e16f67e5a740","resolution":{"observed_at":"2026-08-06T23:50:24.495762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00240","last_updated":"2023-09-30T03:27:31Z","snapshot_observed_at":"2026-08-04T22:51:25.150068Z","submitted_at":"2023-09-30T03:27:31Z","title":"Learning Mask-aware CLIP Representations for Zero-Shot Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00240","snapshot_observed_at":"2026-08-06T23:50:23.369382Z","title":"Learning mask-aware clip repre- sentations for zero-shot segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.369382Z"},"links":{"cited_paper":"/paper/2310.00240","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:16567e05d71ef52d70ee17a0bed8f1b734993a08ebbdc3fdfdf183261a0ed845","observation_id":"8e05eb0c-bd88-40b9-a4bf-6d017e7c101b","resolution":{"observed_at":"2026-08-06T23:50:23.369382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.475008Z","title":"Collaborative vision-text rep- resentation optimizing for open-vocabulary segmentation","venue":null,"work_id":"4fa75ffb-a5a5-4ae4-beca-77423c9d00cc","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.374458Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:7eb946a8d3cd6604f45976357adc4f6047ed4a901b935e7fec1f9ddb2ccd676b","observation_id":"1fc46008-179d-4f55-97b9-d16c91fbdb51","resolution":{"observed_at":"2026-08-06T23:50:24.479936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.459136Z","title":"Weinberger, Serge J","venue":null,"work_id":"35f2523b-ccfc-4486-b656-d412dd3d921a","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.379399Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:fb4fa8931a3bf32ca567d8887440468a037d98487a069d803524c4e6d406e4f0","observation_id":"498432c4-95c4-45e0-855e-1e3a1fe63fd2","resolution":{"observed_at":"2026-08-06T23:50:24.464721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.443324Z","title":"Unicoder-vl: A universal encoder for vision and lan- guage by cross-modal pre-training","venue":null,"work_id":"222f19c9-898d-44eb-bd59-ad8e348981be","year":2020},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.384236Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:d270b78d31bc8b2bc5528b12c6ccb7b3b9726985230327ee8795aee0feffb25f","observation_id":"0a889458-a699-4ddb-8b2b-e8528c40033f","resolution":{"observed_at":"2026-08-06T23:50:24.448213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.427053Z","title":"Ordinalclip: Learning rank prompts for language-guided ordinal regression","venue":null,"work_id":"fb6117ec-986f-4985-bb3d-4b5d302a877a","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.388956Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:5bcb0c97ec65879ec898e254958e40f42ef954ebde8a6cf6698fca0010537427","observation_id":"0ebd4417-b1aa-4019-927d-55a82dc79e9f","resolution":{"observed_at":"2026-08-06T23:50:24.432092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.411347Z","title":"Oscar: Object-semantics aligned pre-training for vision-language tasks","venue":null,"work_id":"2844ab83-fe87-4e64-af11-bf642b407a84","year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.393647Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:763d14547a89b172e7cc0f3b1622f676286f908fa4558940048f2529c3e53abb","observation_id":"c811938a-aac7-4ac5-8e09-acf2ef80452c","resolution":{"observed_at":"2026-08-06T23:50:24.416066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.04150","last_updated":"2023-04-01T19:00:47Z","snapshot_observed_at":"2026-07-06T14:02:40.989915Z","submitted_at":"2022-10-09T02:57:32Z","title":"Open-Vocabulary Semantic Segmentation with Mask-adapted CLIP","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.04150","snapshot_observed_at":"2026-08-06T23:50:23.398271Z","title":"Open-vocabulary semantic segmentation with mask-adapted CLIP","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.398271Z"},"links":{"cited_paper":"/paper/2210.04150","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:445f8613af527891f56fee4338b7995a39d2f6e62ff601689ae830c614a6f152","observation_id":"68a2fe49-d63e-45c2-bf79-9736b651e201","resolution":{"observed_at":"2026-08-06T23:50:23.398271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.395676Z","title":"Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll ´ar, and C","venue":null,"work_id":"abede16f-a3b0-4153-9517-b74bdefba68b","year":2014},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.402939Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:aa7055f64f13ca2bbc04ac250f2e33435bf0475055781c61e0c6f0e64416bfe1","observation_id":"fd7982dc-77c4-4d1e-ab7a-7ab4b057b9c1","resolution":{"observed_at":"2026-08-06T23:50:24.401125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.379979Z","title":"Quality- aware and selective prior enhancement memory network for video object segmentation","venue":null,"work_id":"9e684047-ed41-4df1-837f-af37a5cf840e","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.407360Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:3ff8f1b9f222d4003e32d2714d7d7dabba1055528d4a0eac8661d0bec0aed317","observation_id":"6efe546e-7a4c-4061-8f09-b8ab3b5e83b7","resolution":{"observed_at":"2026-08-06T23:50:24.385247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.365441Z","title":"Global spectral filter memory network for video object segmentation","venue":null,"work_id":"2a8a18d8-199f-4e97-aaf6-d1dcb8cc2890","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.411715Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:055ef4d2ad1643cff7fa8a8d9aca90a1d52dfb3a371f6eeeb6a736d618236959","observation_id":"66b8b1e0-6604-4c24-998d-108842b10359","resolution":{"observed_at":"2026-08-06T23:50:24.370046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.349863Z","title":"Learning quality-aware dynamic memory for video object segmentation","venue":null,"work_id":"351e3ea7-12b3-4270-b3ca-84a219e1a1e1","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.416391Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:a2c0552adf4b2be31aedf0b812dc16cf8d94c61058c92b8bd16705cdabce2ad9","observation_id":"05c253fb-20d2-474e-b290-67a9a5a52fb9","resolution":{"observed_at":"2026-08-06T23:50:24.355059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01623","last_updated":"2024-11-26T14:03:51Z","snapshot_observed_at":"2026-07-06T16:56:21.561513Z","submitted_at":"2023-12-04T04:47:48Z","title":"Universal Segmentation at Arbitrary Granularity with Language Instruction","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.01623","snapshot_observed_at":"2026-08-06T23:50:23.420753Z","title":"Universal segmentation at arbi- trary granularity with language instruction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.420753Z"},"links":{"cited_paper":"/paper/2312.01623","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:797163bd2890c78f6f844badcd5db0bb85c4a312c44eb854d61387e57dbac775","observation_id":"4568aa7d-2900-49b7-a87c-311f3614fb37","resolution":{"observed_at":"2026-08-06T23:50:23.420753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.334546Z","title":"Open-vocabulary segmentation with semantic-assisted calibration","venue":null,"work_id":"e1983188-589b-4ed4-9986-c40d99568515","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.426153Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:a669056a60005875bcd1a9d9aa37a017a7a64495d369fec43ddf6896e5fe4e33","observation_id":"baedb819-8184-4044-9af5-e317f128b7de","resolution":{"observed_at":"2026-08-06T23:50:24.340256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.320027Z","title":"Learning high-quality dynamic memory for video object segmentation","venue":null,"work_id":"ecd9e43d-39b0-446c-af13-0d6d73b2c4b2","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.430987Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:771625bbf736459c89e9300303a2a3d9e74d9ce3caec6c9ce4bbcb0b070f613e","observation_id":"0b47705e-28cb-468b-a303-fe33b669b1b1","resolution":{"observed_at":"2026-08-06T23:50:24.324665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07062","last_updated":"2023-12-14T03:28:49Z","snapshot_observed_at":"2026-07-06T17:00:16.291803Z","submitted_at":"2023-12-12T08:30:09Z","title":"ThinkBot: Embodied Instruction Following with Thought Chain Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07062","snapshot_observed_at":"2026-08-06T23:50:23.435750Z","title":"Thinkbot: Embodied instruction fol- lowing with thought chain reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.435750Z"},"links":{"cited_paper":"/paper/2312.07062","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:72950e0530b49c05aabbb65af37f0c9cc915d367ab8add1eb8b4e1a4c79d79db","observation_id":"4358f59c-72f4-4cdf-905f-4afcf8fea044","resolution":{"observed_at":"2026-08-06T23:50:23.435750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.304739Z","title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","venue":null,"work_id":"04cd0639-a499-47fc-afe0-570cd6f6f343","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.440778Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:7e7a808265f2db09fd64045a5bd50828d2c372696b941a95566e67c3880af605","observation_id":"1b876d93-17fc-45af-9517-fa78d3341338","resolution":{"observed_at":"2026-08-06T23:50:24.310125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17011","last_updated":"2023-05-26T15:13:44Z","snapshot_observed_at":"2026-07-06T15:34:01.658872Z","submitted_at":"2023-05-26T15:13:44Z","title":"SOC: Semantic-Assisted Object Cluster for Referring Video Object Segmentation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17011","snapshot_observed_at":"2026-08-06T23:50:23.445973Z","title":"Soc: Semantic-assisted object cluster for referring video object segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.445973Z"},"links":{"cited_paper":"/paper/2305.17011","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:f99772760d1f894b7dbd824dad4b7c2bb0dcf130dfbafe574afb46338eb2b8e9","observation_id":"08c7c79b-2c56-4764-b50f-6073c683c43e","resolution":{"observed_at":"2026-08-06T23:50:23.445973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.15658","last_updated":"2024-11-25T17:14:20Z","snapshot_observed_at":"2026-07-06T18:19:25.869339Z","submitted_at":"2024-05-24T15:53:59Z","title":"CoHD: A Counting-Aware Hierarchical Decoding Framework for Generalized Referring Expression Segmentation","version":2},"cited_work":{"arxiv_id":"2405.15658","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.15658","snapshot_observed_at":"2026-08-06T23:50:23.690771Z","title":"CoHD: A Counting-Aware Hierarchical Decoding Framework for Generalized Referring Expression Segmentation","venue":"cs.CV","work_id":"a0781c56-5aff-49cb-a46c-d5cfd7b8a9ac","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.450624Z"},"links":{"cited_paper":"/paper/2405.15658","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:e872197a010b9f9f4e39ce36ced26eb58ed0eea9e87b14442b261339575a42cc","observation_id":"3ba8fd79-3fbd-4731-bda1-b25a15a2fbd1","resolution":{"observed_at":"2026-08-06T23:50:23.696452Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.289079Z","title":"Matrix analysis and applied linear algebra","venue":null,"work_id":"734db809-9369-4e2d-85b9-22a4af48d522","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.456174Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:3f825e5c6e3556ee717ac6481585d512f189dc82a14aed340993adc563f71db6","observation_id":"881002cc-da08-4e94-9b36-86b82af250f6","resolution":{"observed_at":"2026-08-06T23:50:24.294036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.274052Z","title":null,"venue":null,"work_id":"d4f48674-d34e-46c5-8a54-561cb5b318cb","year":2014},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.460868Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:081db866e5ca1feb74d8183ccf0d13ac18d281a38c666e9500d8b6deffd51b20","observation_id":"92649031-db64-4c2c-9ab0-735192d697d6","resolution":{"observed_at":"2026-08-06T23:50:24.279433Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.259295Z","title":"Siri: A simple selective retraining mechanism for transformer-based visual grounding","venue":null,"work_id":"d8fe4c4e-c515-44b1-91af-c0d64b814bc4","year":2022},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.465327Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:345bf39ae1cfad6527a0dbb33e905c93396db3b57a4ce0efc984cfd2647dcbe9","observation_id":"72488162-0515-4ac4-886e-9cf16112dca5","resolution":{"observed_at":"2026-08-06T23:50:24.264057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.243614Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"a774d58c-b451-4d72-aa55-7218356204ad","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.470029Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:e0f13fa3b01274060b894a71f9a0aefc2dcc9858ef61f2bcd6cd0e2533ef5720","observation_id":"f6eb4e40-6224-4237-b50d-b3d76dde49b8","resolution":{"observed_at":"2026-08-06T23:50:24.248952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00603","last_updated":"2024-12-15T12:04:12Z","snapshot_observed_at":"2026-07-06T18:39:03.315371Z","submitted_at":"2024-06-30T06:08:12Z","title":"Hierarchical Memory for Long Video QA","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00603","snapshot_observed_at":"2026-08-06T23:50:23.474555Z","title":"Hierarchical memory for long video qa","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.474555Z"},"links":{"cited_paper":"/paper/2407.00603","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:51b780be1a6b28e6fed82a85d1ba6cafe94082d2ae974d452af08421a8e23c03","observation_id":"ccf2a863-c91f-49d6-a381-782b87cdf801","resolution":{"observed_at":"2026-08-06T23:50:23.474555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.228712Z","title":"Uni-adafocus: Spatial- temporal dynamic computation for video recognition","venue":null,"work_id":"593da257-482e-49e0-b78a-04eb89102515","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.479341Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:66fe6dba8b0406c88a31a0a2adb421d231470fba2ec3a523edba81499606bbc2","observation_id":"191b1b35-0da6-4e33-8a0d-247b3a060ea4","resolution":{"observed_at":"2026-08-06T23:50:24.233510Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.212651Z","title":"Iterprime: Zero-shot referring image segmen- tation with iterative grad-cam refinement and primary word emphasis","venue":null,"work_id":"1c81bdef-808e-4b8a-a91c-2b5baa947f05","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.484430Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:2107142ac7d9ab50ddcbd13ff217c6607ddd8d7118434484a9847cbf26b6f885","observation_id":"8be1fb6b-ddb6-4466-92f6-3e7cc8b144e0","resolution":{"observed_at":"2026-08-06T23:50:24.218075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.197985Z","title":"Sam2-love: Segment anything model 2 in language- aided audio-visual scenes","venue":null,"work_id":"0c8d5337-61f4-4f56-9da0-c7032b2e5b7f","year":2025},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.489538Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:ff5e116a2478ebc7dfb71505b92d3e3218b9905abb73d0e84ae7b7a9940f817a","observation_id":"cb973aa9-23da-4116-afe2-42bef671ea2b","resolution":{"observed_at":"2026-08-06T23:50:24.202554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17606","last_updated":"2024-12-02T03:19:04Z","snapshot_observed_at":"2026-08-06T16:07:46.683644Z","submitted_at":"2024-11-26T17:18:20Z","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17606","snapshot_observed_at":"2026-08-06T23:50:23.494720Z","title":"Hyperseg: Towards univer- sal visual segmentation with large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.494720Z"},"links":{"cited_paper":"/paper/2411.17606","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:e5341549fba4388ee1325fc85efa4c68c8af4c8f889a2b142d6b184b6cfd7739","observation_id":"494fb9ba-fec2-4b55-8b91-96c449cbf13d","resolution":{"observed_at":"2026-08-06T23:50:23.494720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14006","last_updated":"2024-12-18T16:20:40Z","snapshot_observed_at":"2026-07-06T20:09:24.415233Z","submitted_at":"2024-12-18T16:20:40Z","title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14006","snapshot_observed_at":"2026-08-06T23:50:23.499385Z","title":"Instructseg: Unifying instructed visual segmentation with multi-modal large lan- guage models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.499385Z"},"links":{"cited_paper":"/paper/2412.14006","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:feeb3599dcb11dfa5e7271f7be0921b38ac3de70e8983c61210a36a0106ee02b","observation_id":"0cb8fa4d-6035-476f-ae78-1a5d6d762ba8","resolution":{"observed_at":"2026-08-06T23:50:23.499385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.182017Z","title":"A large-scale benchmark for food im- age segmentation","venue":null,"work_id":"b9134be0-e140-4828-af8a-52ca84c4e97d","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.504013Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:cef050e766dbe3d7052c74a34ed942241f7950c221eb9426522345613dd0e274","observation_id":"6c4d208a-bab4-40b6-90d7-116debeb9aed","resolution":{"observed_at":"2026-08-06T23:50:24.187846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.165622Z","title":"Detectron2","venue":null,"work_id":"4cd24c6a-087e-4be6-96a7-3a8f1a66c367","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.508334Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:3a4d9fa158047881805c208841e508603063a67258460b3c4d1b12c6e2e39daf","observation_id":"e17d219c-1cdf-4ab9-a8b5-4882523f970d","resolution":{"observed_at":"2026-08-06T23:50:24.170493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.150716Z","title":"Semantic projection network for zero- and few-label semantic segmentation","venue":null,"work_id":"47ab7dd9-8d6d-4b92-ac13-01e0336e758e","year":2019},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.512584Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:782421f94e3b5a5274f21d9aaf70470e70aae9f0dd06e532e04023931185be9a","observation_id":"6f439054-5bbb-428e-959b-d363152575d9","resolution":{"observed_at":"2026-08-06T23:50:24.155465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.134673Z","title":"Bridging the gap: A unified video comprehension framework for moment retrieval and highlight detection","venue":null,"work_id":"f003df54-9ad4-4f7e-95db-32a7d7ee7492","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.516744Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:3e31530375b9ff15cc080e93884ebf9b10719765576604eb23f3b77e8b6e9e4a","observation_id":"8541a39c-81cf-4695-a0d3-dce7b50db631","resolution":{"observed_at":"2026-08-06T23:50:24.140251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.119024Z","title":"Sed: A simple encoder-decoder for open- vocabulary semantic segmentation","venue":null,"work_id":"d9557ef1-89dd-45cd-8c03-ba970511aa9c","year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.521827Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:6aa0e1e1349f03433dd50b65a48c5a8decfd81af37e90fa9b6b0788ab764a7e8","observation_id":"a8f54fe3-8ea2-4cab-997b-ab0e6d579725","resolution":{"observed_at":"2026-08-06T23:50:24.124526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.105131Z","title":"Alvarez, and Ping Luo","venue":null,"work_id":"e184d97d-0173-4501-b226-0454e19223ac","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.526934Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:1be25b8d764f3a33592e80022fe21589928c9f6afcece5598bec061de0571686","observation_id":"fe3547b4-b4dd-4f37-83c1-97cfd2ad475c","resolution":{"observed_at":"2026-08-06T23:50:24.109620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.088706Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":"cbbf586e-c63a-48a2-9c1e-a46203834090","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.531413Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:c9f97dd082034e44836cf68b42d53be0287dff5c7cd483ad19fc6b8b65e839b3","observation_id":"fe505afc-f573-406b-8518-9bbc306c6719","resolution":{"observed_at":"2026-08-06T23:50:24.094278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.14757","last_updated":"2022-12-29T16:36:55Z","snapshot_observed_at":"2026-08-06T08:59:22.727151Z","submitted_at":"2021-12-29T18:56:18Z","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","version":2},"cited_work":{"arxiv_id":"2112.14757","doi":null,"metadata_source":"pith","pith_arxiv_id":"2112.14757","snapshot_observed_at":"2026-08-06T23:50:23.619666Z","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","venue":"cs.CV","work_id":"ec5226e2-f2be-4f97-b540-beabba9fed52","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.536291Z"},"links":{"cited_paper":"/paper/2112.14757","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:5412336738263604b200e38180d91cfe52cdb2dcca1b7dbce15610a645991f37","observation_id":"17813763-b4c7-4cc1-a2ea-c59571aa80ca","resolution":{"observed_at":"2026-08-06T23:50:23.626640Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.072226Z","title":"Side adapter network for open-vocabulary semantic segmentation","venue":null,"work_id":"fd311380-a784-46ec-84d8-4ceb2f3cff08","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.541877Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:5af1a36b42595b38824686cfd179e971cc6cadf54c0ce2f3dfe3b2f6923e66e6","observation_id":"cd68d2cc-bf5d-4dde-bcd9-7e618daee1d8","resolution":{"observed_at":"2026-08-06T23:50:24.077918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.056926Z","title":"Masq- clip for open-vocabulary universal image segmentation","venue":null,"work_id":"5fb193d3-438d-4df5-9008-ed0274d5c8d0","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.546341Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:1957897ddf8aa7c67791e8b52f2681bbc50da15add401b2af740253f340736a4","observation_id":"d0e369b9-683d-46a4-8293-042169a04220","resolution":{"observed_at":"2026-08-06T23:50:24.061442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.040567Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip","venue":null,"work_id":"e4128890-7dde-49ad-ad0f-ccde289b506f","year":2023},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.550790Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:6a38c52e6c2cb7e40df2bcfe69548237084b03af8f4f1c4087d197e58f3f4413","observation_id":"bd2b2bcc-c07e-4916-84ae-123c09abd7b5","resolution":{"observed_at":"2026-08-06T23:50:24.046307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:24.018215Z","title":"Prototypical matching and open set rejection for zero-shot semantic segmentation","venue":null,"work_id":"295f824b-d1c0-43bf-a773-9846266de0c5","year":2021},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.555457Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:74bbd7a6116402834b9ecf76c4e6e65c8c688d20e77618513bc995b2741bfeb8","observation_id":"0e8d8dd4-b8d1-4016-b6c3-8bb89b40456c","resolution":{"observed_at":"2026-08-06T23:50:24.024982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-06T23:50:23.560907Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.560907Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:990990e2a0bb18ca069e75b1eccb92e192812f37ac4299ac31e16f7ceb5c1bc1","observation_id":"fe2be44c-1e16-44a6-ad2d-f8e543fb0667","resolution":{"observed_at":"2026-08-06T23:50:23.560907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:50:23.993862Z","title":"Scene parsing through ADE20K dataset","venue":null,"work_id":"5678b5eb-6a93-46c1-a5e9-e20928464ce5","year":2017},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.565560Z"},"links":{"citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:b4efc1073b411c13704ba7079e22a3da23b8e644e18e54372a92fe40a53f347f","observation_id":"d2d361ab-ea73-4d7a-9d59-648b450a27b6","resolution":{"observed_at":"2026-08-06T23:50:24.002543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":3,"verified_fuzzy":43},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 0 inbound Pith citation observations for arXiv:2506.16058."}