{"as_of":"2026-08-17T06:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f1552b93a91120fd3f31a71653bcd0b91deaa5f5c9e57fb94fb70e6853c62806","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T17:11:26.749841Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.09329/citation-record","integrity":"/paper/2412.09329/integrity","json":"/paper/2412.09329/citation-record.json","paper":"/paper/2412.09329"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.421950Z","title":"Mining contextual information beyond image for semantic segmentation,","venue":null,"work_id":"790fcf6c-6b38-497f-8904-1c1f4d8349cd","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.523458Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:27447b25bf1b9ebfed91efbea325f34f185ab088ae07d58e8661ef673ff96b43","observation_id":"81e86c29-4f9d-4375-a079-dae5ecb5d4d8","resolution":{"observed_at":"2026-08-11T17:11:27.427742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.410323Z","title":"ISNet: Integrate image-level and semantic-level context for semantic seg- mentation,","venue":null,"work_id":"de53ee17-30c2-4acd-9ddb-36e9dbab7dfa","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.527818Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:9d8daf306b15f9e8ccbd56f749ef8a53af2c2ffa9c10373865e0d7631d80760d","observation_id":"02f48fa2-06de-4412-b765-e1057c8f8ada","resolution":{"observed_at":"2026-08-11T17:11:27.414580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.398729Z","title":"Boundary-guided lightweight semantic segmen- tation with multi-scale semantic context,","venue":null,"work_id":"5d6ae8d8-ce26-4623-ab95-516a505b71e1","year":2024},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.531421Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:000d746abbdd237c6b81f6c817e3b7507161b8ad703a52e25e0aa3641767661c","observation_id":"7a96571f-faa7-4855-859d-a541140e2086","resolution":{"observed_at":"2026-08-11T17:11:27.402676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.386182Z","title":"Fbsnet: A fast bilateral symmetrical network for real-time se- mantic segmentation,","venue":null,"work_id":"d9164f36-7315-41be-9915-ab71dcee7fe4","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.535174Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:b0209ef55647d878eb888c3d7beeb78b380a1c64808facbe75b4259540b6074b","observation_id":"c3033095-cafe-43cb-8909-59150ceef491","resolution":{"observed_at":"2026-08-11T17:11:27.391161Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.374337Z","title":"Semantic segmen- tation guided pixel fusion for image retargeting,","venue":null,"work_id":"974fc43c-98e8-4d8d-821d-922de977556b","year":2019},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.539049Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:baf36912ddded0a57a762732cfc6b292706ea32b6e3a5cbe48fee357f23b4ebd","observation_id":"ad32c5fe-0071-4c61-bfb8-ae549fd86b3a","resolution":{"observed_at":"2026-08-11T17:11:27.378207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.363730Z","title":"Difference-aware distillation for semantic segmenta- tion,","venue":null,"work_id":"8b7e4744-fee7-4c0f-8df7-cd744cee56ff","year":2024},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.542589Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:5eff2f3f7e64aca7e0f9a2a3bd02a5210ad13e388d8aec9a015e98fb000d2218","observation_id":"9489911a-7756-4592-bdcf-4b2c6b00ea9f","resolution":{"observed_at":"2026-08-11T17:11:27.367388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.352477Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"a6830efa-d9be-4264-b20c-c96d1262f7f3","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.546378Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:17341f9f71af42d9f293f8d7bf2a73249364afd4cb142c2649fc0b64796ce77c","observation_id":"1a665bc3-41a8-4314-b182-8101cea5ea32","resolution":{"observed_at":"2026-08-11T17:11:27.356619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.340932Z","title":"Side adapter network for open-vocabulary semantic segmen- tation,","venue":null,"work_id":"59548f25-66ea-41e2-9030-42e70c5df9d9","year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.549781Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:a7994ea5f4eb67cd53fe551c2ee73a6480630a1fdecae66378a126dc6175a2ae","observation_id":"92b5a785-89f4-4373-8e9b-9b3cd9cfefd9","resolution":{"observed_at":"2026-08-11T17:11:27.344907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.329226Z","title":"Scaling open- vocabulary image segmentation with image-level labels,","venue":null,"work_id":"79c0bf6b-9d03-4668-aebf-8ba4d674f040","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.553084Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:dac284aa452aefab8d83fca9bf61c039650ec880d5247fc1c3ffb2442349e973","observation_id":"52e28d88-2909-48a0-ab9a-0f60cb700556","resolution":{"observed_at":"2026-08-11T17:11:27.333113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.317472Z","title":"Towards open-vocabulary video instance segmentation,","venue":null,"work_id":"6e989e79-25e1-45f3-bd83-822610065c5b","year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.556384Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:9924db4b1bfd95c202648f6aa41d5d50c72ac50a9f75d2a3c2fdefb6763e98e6","observation_id":"dbfa2416-1278-46c3-98df-369ad52ff9e7","resolution":{"observed_at":"2026-08-11T17:11:27.321825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16835","last_updated":"2024-08-17T09:30:31Z","snapshot_observed_at":"2026-08-16T23:34:48.842156Z","submitted_at":"2023-05-26T11:25:59Z","title":"OpenVIS: Open-vocabulary Video Instance Segmentation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16835","snapshot_observed_at":"2026-08-11T17:11:26.560061Z","title":"Openvis: Open-vocabulary video instance segmentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.560061Z"},"links":{"cited_paper":"/paper/2305.16835","citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:54a5ba481b38e69e0799786a077ad3cf954616ed239e3860563c7ffa6c71bc11","observation_id":"97bd1120-1f65-421a-8216-a681559d0913","resolution":{"observed_at":"2026-08-11T17:11:26.560061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.306349Z","title":"VSPW: A large-scale dataset for video scene parsing in the wild,","venue":null,"work_id":"d61f2901-1c16-45b6-a5ab-2812e8c7e6b2","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.563955Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:287f9a7ded5169da496439522c248358a86489023df6eea252faf4d86a4f7d23","observation_id":"d61852ce-00dd-40ff-a977-e673ca7de199","resolution":{"observed_at":"2026-08-11T17:11:27.309997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.295562Z","title":"The Cityscapes dataset for semantic urban scene under- standing,","venue":null,"work_id":"b2744e7b-5c3c-4131-904d-960b0ccc43ca","year":2016},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.567553Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:0cf695bb4a07b7abc8dba3650850f78e0b3662c6ae51727de11903b7f938abac","observation_id":"282073e0-4599-4ea9-8e45-9c2a8ec6390b","resolution":{"observed_at":"2026-08-11T17:11:27.299331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.284529Z","title":"In- door segmentation and support inference from RGBD images,","venue":null,"work_id":"845d223a-c2c8-4515-a508-2918405536bc","year":2012},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.570822Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:3eb253e4bb6357662d738c521543943fd902e0d7b993bf027894de9b5721199e","observation_id":"4824f420-a459-4b02-acc6-a5a73e959148","resolution":{"observed_at":"2026-08-11T17:11:27.288361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.273502Z","title":"Segmentation and recognition using structure from mo- tion point clouds,","venue":null,"work_id":"be6d79b9-9a5b-4a90-81df-35294719688a","year":2008},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.574234Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:447636ed4384ab8f48ff5982812a7f22f5caf94011caca1c35e9ee5293d0cfd5","observation_id":"8d4b902f-e5ac-4b23-806b-ec93d8658613","resolution":{"observed_at":"2026-08-11T17:11:27.277157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.262583Z","title":"Low-latency video semantic segmentation,","venue":null,"work_id":"5e1c0dd4-8afe-40d4-9292-33f6ea7285a4","year":2018},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.578371Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:62c30f823dd522c24f844251f362aa914ea8456a14472f7de31a6d47496d7066","observation_id":"dd886d01-766c-44a3-9af8-060cad5a563c","resolution":{"observed_at":"2026-08-11T17:11:27.266496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.252121Z","title":"Clockwork convnets for video semantic segmentation,","venue":null,"work_id":"43820380-640e-403d-b3f0-b39c4b574a1b","year":2016},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.582186Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:7cabd29148272fb6606e5f70b654c99893d402267eda3b79973bc86645b0d565","observation_id":"2fafca02-3be7-48f2-b584-ec1f41f3f764","resolution":{"observed_at":"2026-08-11T17:11:27.255692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.241390Z","title":"Budget-aware deep semantic video segmentation,","venue":null,"work_id":"c7075783-3cc8-44c4-a167-1cfa5c9e00d8","year":2017},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.585829Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:8df8bf3a8748756c9fea9d613b8f27f0f2dc144fd311d3edbd6049fb3e8b79e0","observation_id":"8971abdd-0c8b-4a85-8b67-4feaba13395d","resolution":{"observed_at":"2026-08-11T17:11:27.245194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.230770Z","title":"Deep feature flow for video recognition,","venue":null,"work_id":"2ebda3f9-b0c7-4aa2-9758-c636cf142315","year":2017},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.589750Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:45a60d6384e0513b8fb84bf5d7a15b419bff1c4f5fa089cef4414f177c6f379d","observation_id":"afa76a1f-43b7-4f35-9d66-3cef598ae2e4","resolution":{"observed_at":"2026-08-11T17:11:27.234774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.220417Z","title":"Dynamic video segmentation network,","venue":null,"work_id":"8fecb931-b97b-4a2d-b546-13ae3eafe88f","year":2018},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.594807Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:37068d08b7fd23ba20e08b02655aed3e317a1fb6cfa8cbd0c6f48574cbab4b0f","observation_id":"6102fd91-2f2c-42d2-980f-19539f5335af","resolution":{"observed_at":"2026-08-11T17:11:27.224028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.209953Z","title":"Accel: A correc- tive fusion network for efficient semantic segmentation on video,","venue":null,"work_id":"713f4038-6170-4b1c-bc01-3ac93e7991b7","year":2019},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.598832Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:3f36948b786e13a2c3568684d37e1af3222875e29c85fed4a07e088f75b3b86b","observation_id":"0ceae0fd-845a-448b-b78d-4e7fdab0667c","resolution":{"observed_at":"2026-08-11T17:11:27.213728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.199308Z","title":"Temporally distributed networks for fast video semantic segmentation,","venue":null,"work_id":"4a0603c3-801e-4164-878c-ac810e2c8b57","year":2020},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.603256Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:d8a040b5b5b1dd4504f006a99e6e608ef8106b23cf7f23d53c5da719ba2de207","observation_id":"e7fb2c42-0d9b-4388-bf31-aa1013de3212","resolution":{"observed_at":"2026-08-11T17:11:27.203054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.188126Z","title":"Efficient se- mantic video segmentation with per-frame inference,","venue":null,"work_id":"d8663efc-ea14-469a-868f-4214d5da0de4","year":2020},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.607013Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:1fdba4fe531d12d3372cbcd46e0e4bafcbbbcbe63143d24d56ebfb76aac3ed0e","observation_id":"d5a48f03-a61f-492e-aa25-2c388d58f4e7","resolution":{"observed_at":"2026-08-11T17:11:27.191657Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.177582Z","title":"Local memory attention for fast video semantic seg- mentation,","venue":null,"work_id":"51807119-ae1f-4a88-9d78-ef8c08aa2561","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.610719Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:5645e2376427775b20103969dad5f2297577e8522de7f71a695f7d9f15f1fa04","observation_id":"7a4c6c64-4370-4914-b0fe-5cced1b01a02","resolution":{"observed_at":"2026-08-11T17:11:27.181279Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.166043Z","title":"Feature space optimization for semantic video segmentation,","venue":null,"work_id":"dddbb022-19bc-4be6-b910-cdd9fa52d1d6","year":2016},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.614271Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:5b0e81338b6838fa3ca7c84fc5bc6df2563d090501de61dd7d1548b252a12433","observation_id":"627b6002-6dcb-4eb8-9502-908c6b054a83","resolution":{"observed_at":"2026-08-11T17:11:27.170003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.155552Z","title":"Video semantic segmentation via sparse tem- poral transformer,","venue":null,"work_id":"559bc31d-e249-4b15-b93c-af6a23ddce1e","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.618146Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:9fa04ccec4116438c268aacd20e41a51743d0d65769a0bf87963047751067e61","observation_id":"93237ddc-f16d-4e40-8aa1-221174f1d06c","resolution":{"observed_at":"2026-08-11T17:11:27.159078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.144792Z","title":"Semantic video CNNs through representation warping,","venue":null,"work_id":"77944d38-7272-4d49-a90f-44f9dfc87c7e","year":2017},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.621869Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:19612742cda8cc7dbea9d45ecba47ee0b0521b2ef54e57f8335f83a8d0edbcea","observation_id":"dde8f402-7a74-4a75-aa1e-c65e96f1099a","resolution":{"observed_at":"2026-08-11T17:11:27.149016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.133883Z","title":"Video scene parsing with predictive feature learning,","venue":null,"work_id":"a4fa114f-e566-48cd-93ce-0f77fdd77e7f","year":2017},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.625700Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:f5fb841f0150773fe3093abc1d71093a0a31290b951978c7c7b7f9cc433b51ff","observation_id":"68fb74be-67ad-4c62-a6c2-f6e687da86dd","resolution":{"observed_at":"2026-08-11T17:11:27.137843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.123310Z","title":"Surveillance video parsing with single frame supervi- sion,","venue":null,"work_id":"dafe71d0-f7ba-4118-84aa-9d10f30b9b22","year":2017},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.629783Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:74a3e717ecdc726487bf045bdc985529adb591a39c3ad58a71435a529288c6b0","observation_id":"c0bc9b50-84cc-47c8-b08e-3e274fafa66a","resolution":{"observed_at":"2026-08-11T17:11:27.127238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.112154Z","title":"Semantic video seg- mentation by gated recurrent flow propagation,","venue":null,"work_id":"f86e1740-cff7-4666-b4e8-4ed6cfb84178","year":2018},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.634349Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:974a7b1e3fc154554c643bfea117fc2cc70ed2c226010f7514efd046a2e3441c","observation_id":"a6e1c533-0069-4787-8762-0c08c3cc94ea","resolution":{"observed_at":"2026-08-11T17:11:27.115836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.101912Z","title":"Improving semantic segmen- tation via video propagation and label relaxation,","venue":null,"work_id":"9036ced0-bd22-4e88-b4bb-93acbe599059","year":2019},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.638204Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:178677b5bfaa6b49cb78c3b3d145c2905f99b43f00cba0f5eba91a4f1ec71110","observation_id":"974c22ee-1672-4d82-80fe-4d68c1df5601","resolution":{"observed_at":"2026-08-11T17:11:27.105727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.091336Z","title":"AuxAdapt: Stable and efficient test-time adaptation for temporally consistent video semantic segmentation,","venue":null,"work_id":"62e698f8-e37a-499c-8c3f-35d1dfffc065","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.642027Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:edd5620bdab1a1911a0d40660c3535f3aa3de12731060579735c266fa1534104","observation_id":"d3c4eea2-d354-4d0b-a792-42882983972e","resolution":{"observed_at":"2026-08-11T17:11:27.095049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.080201Z","title":"Coarse-to-fine feature mining for video semantic seg- mentation,","venue":null,"work_id":"1b0a1309-2991-479a-9a73-dfbba8aa020b","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.645691Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:0c184496a857755b682905cfbd8d6f52258f3aa38cfc51fbe531104498d5ef9e","observation_id":"bae41cb6-1f9a-4a52-a2cf-a08c6b0bba00","resolution":{"observed_at":"2026-08-11T17:11:27.084007Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.069201Z","title":"Mining relations among cross-frame affini- ties for video semantic segmentation,","venue":null,"work_id":"5639c370-caf7-4947-987f-36c69306b9ce","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.649497Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:bcb6b82b924684aecb479990e0d71d626208a44183450aa20dd1cd044e4e67df","observation_id":"9fa7ab79-35c9-46c2-b22a-146ef82a19de","resolution":{"observed_at":"2026-08-11T17:11:27.073163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.057851Z","title":"Learning local and global temporal contexts for video semantic segmentation,","venue":null,"work_id":"0340ff45-5e71-405d-8ec6-6095ae372bf9","year":2024},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.653524Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:0d26cd52537d649640b1075b736f1febca180c5a092486f80cd0b3764df1992a","observation_id":"319d3c95-ce45-448a-a97a-3dcd08a8c798","resolution":{"observed_at":"2026-08-11T17:11:27.061996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.046048Z","title":"Video K-Net: A simple, strong, and unified baseline for video segmentation,","venue":null,"work_id":"55ac8912-9095-4b8f-813e-cc4810560e4f","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.658184Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:66dad8f847eef5c39fccf34cbebe36f4e1f910a00075a94213f1e12a8c5aa6ad","observation_id":"26e07b58-8010-41b1-bd01-809b9d85c177","resolution":{"observed_at":"2026-08-11T17:11:27.049959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.034777Z","title":"Tube-Link: A flexible cross tube baseline for universal video segmentation,","venue":null,"work_id":"0161b4c5-bd30-4a7a-b736-de219682bc81","year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.661899Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:b9ee2898e1b2bdd9e7de644deabc5d265cd08d8df28c19f475a6603b7e1cfd08","observation_id":"86cac5d9-1dd0-430d-ada2-cd7461dafbec","resolution":{"observed_at":"2026-08-11T17:11:27.038709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.022343Z","title":"Mask propagation for efficient video seman- tic segmentation,","venue":null,"work_id":"eef1a507-9409-48ff-b897-5a68106098a2","year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.666155Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:f3d27c46f77d883b57f4d7b422ef1547f9c5f9a65176f0546ec7b8e32eeb24d9","observation_id":"5ebbae45-4990-4a75-9983-7a924569947e","resolution":{"observed_at":"2026-08-11T17:11:27.026447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:27.009707Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision,","venue":null,"work_id":"07a39b6f-858a-4151-b8ac-65068dc8f1df","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.670412Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:67b9625be3f7bc5fa50bb19fb17a1342b9188b401d5dc026b5edac8aef6dc447","observation_id":"66cedfee-721d-4778-ac7a-aec0adae826e","resolution":{"observed_at":"2026-08-11T17:11:27.013685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.674080Z","title":"Decoupling zero-shot semantic segmentation,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.674080Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:31e9f000157c5c7818ea8732be059efcd48eee2e2d8e0e868efc0b09e9934e0b","observation_id":"1bdc92d7-7239-468d-8113-5974d44d47e1","resolution":{"observed_at":"2026-08-11T17:11:26.674080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.992088Z","title":"P2T: Pyramid pooling transformer for scene understanding,","venue":null,"work_id":"a3601ea5-aa4f-40cb-b04b-b4b58532b269","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.677839Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:e548114fe56ca5c4a5f33509399881656313cad4ffbde100b9e76b30f1124041","observation_id":"4c9221c0-6dc8-428d-8271-a52fcf48f837","resolution":{"observed_at":"2026-08-11T17:11:26.995835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.981440Z","title":"Object-contextual rep- resentations for semantic segmentation,","venue":null,"work_id":"08582e53-659d-4b6c-9310-680c9814b9c2","year":2020},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.681230Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:2d38e76492ab0c9fc7fa0d998d97699497079a1061d399d2f8deb366b1222488","observation_id":"db07c70f-059a-4257-85fd-e52b1514ada0","resolution":{"observed_at":"2026-08-11T17:11:26.985143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.969231Z","title":"Attention is all you need,","venue":null,"work_id":"10e9cf2f-92a4-4bc3-8a37-863d30f6e606","year":2017},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.684503Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:9ae4147a7e0d2dceab638395953a66f4cec4d25c45fff73825abc04aa1307509","observation_id":"28507564-4460-4729-9720-ad8a0df8f6ff","resolution":{"observed_at":"2026-08-11T17:11:26.973924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11797","last_updated":"2024-03-31T11:53:55Z","snapshot_observed_at":"2026-08-16T15:46:34.789160Z","submitted_at":"2023-03-21T12:28:21Z","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11797","snapshot_observed_at":"2026-08-11T17:11:26.688063Z","title":"CAT-Seg: Cost aggregation for open-vocabulary semantic segmentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.688063Z"},"links":{"cited_paper":"/paper/2303.11797","citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:e415cf2c478bd6f7138bfbeb161e44b55ff17790dbf0bcff2f177204145d221c","observation_id":"aa28fe8f-a11d-441a-abdd-de22ef3fe6bc","resolution":{"observed_at":"2026-08-11T17:11:26.688063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.691746Z","title":"Open- vocabulary semantic segmentation with decoupled one- pass network,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.691746Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:699eed940f3cf8cd7f4ec3bd176ee882109d6f3896b410c4a2cb186f6a0cb470","observation_id":"0028ff7d-c385-4069-85b8-3d9b94ff78eb","resolution":{"observed_at":"2026-08-11T17:11:26.691746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.951495Z","title":"A simple baseline for open-vocabulary semantic segmentation with pre-trained vision-language model,","venue":null,"work_id":"ab862da2-8af3-4353-bbc0-1932d4439ce8","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.695206Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:7ef054062d23012971049d7094a9c7a5041cd1f3077942bff8e654537b0039d5","observation_id":"0557ca17-5af9-486d-bbe7-079b0499f56c","resolution":{"observed_at":"2026-08-11T17:11:26.955341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.940268Z","title":"FreeSeg: Unified, universal and open-vocabulary image segmentation,","venue":null,"work_id":"d6ac9c47-46fa-4b3b-8976-1f098b8294f1","year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.698823Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:d095ef08e31b76f0f3c774303baf5ef7b68e74b9a3d39e4e4aa1e51e558e155d","observation_id":"f977d459-4636-4c00-bba0-451543cccecd","resolution":{"observed_at":"2026-08-11T17:11:26.944054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.928184Z","title":"Segment anything,","venue":null,"work_id":"e5ccef92-d38e-4d42-8a12-a1ca9fd837fb","year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.702201Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:7bfbc040537c3cc86d43d33425f720e648defde297db09ecdbe1673a5502797d","observation_id":"9aa7eb27-0aee-4dd2-a39b-ff521d0c6ac8","resolution":{"observed_at":"2026-08-11T17:11:26.932089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.916961Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale,","venue":null,"work_id":"f67993de-2049-4d12-a4bc-6584fdda9a5c","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.706197Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:d98b84efbb636fa71420911e9dfc8617e97ed5c6f3b6f7cd345f6a549d193b60","observation_id":"16387f5a-0b5d-4845-ac12-08d25bc9cf07","resolution":{"observed_at":"2026-08-11T17:11:26.920788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.710110Z","title":"Deep residual learning for image recognition,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.710110Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:fbe127b8b39a5c0f581388266bfb23b6720d95b2fa838f79bc57ecaeca4407bf","observation_id":"8c8fbd71-1638-4165-8c6f-6bc1c68a14f2","resolution":{"observed_at":"2026-08-11T17:11:26.710110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.898902Z","title":"Adam: A method for stochas- tic optimization,","venue":null,"work_id":"4028f450-2240-4d45-8a49-415f66a552ea","year":2014},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.714856Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:4f5cb24811848400f80d1f7cbdb872da2f52244baddce92bdd53e4e436eb93ee","observation_id":"0497bd56-09bb-41fb-ab76-7a9f35eade14","resolution":{"observed_at":"2026-08-11T17:11:26.903013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.886420Z","title":"Deep-irtarget: An automatic target detector in infrared imagery using dual-domain feature extraction and allo- cation,","venue":null,"work_id":"f84cf9ba-969e-4b6c-ac37-52ee80c49805","year":2021},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.719034Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:c43e1aaa1f658d2171d3132ad4380c23af8a74e6473a34f2086002442cf55ce4","observation_id":"5fdec607-f165-4b1d-a85e-0b71df270525","resolution":{"observed_at":"2026-08-11T17:11:26.891752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.874397Z","title":"Semantic scene completion from a single depth image,","venue":null,"work_id":"27ec0abc-97a3-4944-988c-ff6ab315e256","year":2017},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.722768Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:a943d90111ac681a16057896790119c3ad10e7f20f8d27b303247f148c0b7ff6","observation_id":"a7ca01f1-4d8c-4ed9-95dd-910ed3c26d0c","resolution":{"observed_at":"2026-08-11T17:11:26.878108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.863279Z","title":"Con- trastive boundary learning for point cloud segmentation,","venue":null,"work_id":"e7f65dbb-abdb-4f01-ab75-4e20cdd69bce","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.726425Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:1c6f8b0eafad5e53baa06d55e3a9e724db8ba09fa6aac58b52649689a1625a9c","observation_id":"1de4735a-cdf3-4795-b0b2-d142ba9224e3","resolution":{"observed_at":"2026-08-11T17:11:26.867046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.850927Z","title":"Learning from noisy labels with deep neural networks: A survey,","venue":null,"work_id":"9a0ab796-584f-4672-91ff-e603c1d8fb76","year":2022},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.730497Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:ccd8bb8bab267b4a201cc240d4909e54dd3f5772f9cd2f3180a75f05f8fb2016","observation_id":"a1f9b628-a1a2-46c7-b622-df85e84be200","resolution":{"observed_at":"2026-08-11T17:11:26.855396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.839628Z","title":"Cognition-driven structural prior for instance- dependent label transition matrix estimation,","venue":null,"work_id":"c3ad2873-bcb2-4fe3-b178-2dddc3120754","year":2024},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.733962Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:b3f082c40db4812b866b1496fb25b1fd122f84bf1713ac89b5ba34b0647bda03","observation_id":"972b680e-8ebe-4456-a322-ca9b6909f5e3","resolution":{"observed_at":"2026-08-11T17:11:26.843706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.827991Z","title":"Feature modulation transformer: Cross-refinement of global representation via high-frequency prior for image super-resolution,","venue":null,"work_id":"69b02e25-835c-495e-b4de-f7bd36a91079","year":2023},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.737738Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:ad0d6e67f04dec022c3094ab8a3b026634e8ca687f4823140bece7fe548ccaa6","observation_id":"3b2d0864-d636-4836-bcc0-ac34a8d7c856","resolution":{"observed_at":"2026-08-11T17:11:26.832274Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.741789Z","title":"Panet: Few-shot image semantic segmentation with prototype alignment,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.741789Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:c174feee19b012c44c4161a27b4316efb4229cc9223d6709c7ff66621619cf36","observation_id":"4e297a28-a321-41b2-969d-9fee4b2c15d5","resolution":{"observed_at":"2026-08-11T17:11:26.741789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.809908Z","title":"Part-aware correlation networks for few-shot learning,","venue":null,"work_id":"b77b2c68-29bf-4ae6-bad6-237b6f934df6","year":2024},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.745666Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:9754b68eb7e59c5a3d1ca963285fde328e072992106a6093e4546e7b83294df9","observation_id":"dd3fc7a2-3f14-4046-8a4c-f174a650ab0b","resolution":{"observed_at":"2026-08-11T17:11:26.814259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:11:26.797689Z","title":null,"venue":null,"work_id":"bc7677c5-fe10-481f-9193-9a74f4db61b1","year":1998},"citing_paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation","version":1},"reference_index":1998,"source":"pdf_text","source_observed_at":"2026-08-11T17:11:26.749841Z"},"links":{"citing_paper":"/paper/2412.09329"},"observation_digest":"sha256:50714d77679eb05aeae7c629b2982871c1299b81d90442fb6a18b41727e1f4cb","observation_id":"77c0a1a2-cea4-43d1-902e-030e8f973aab","resolution":{"observed_at":"2026-08-11T17:11:26.802409Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.09329","last_updated":"2024-12-12T14:53:16Z","latest_version":1,"primary_category":"cs.MM","snapshot_observed_at":"2026-08-11T17:03:23.530063Z","submitted_at":"2024-12-12T14:53:16Z","title":"Towards Open-Vocabulary Video Semantic Segmentation"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":53},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2412.09329."}