{"as_of":"2026-08-12T15:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7cbbb71742a1896f93fd660fe82ccc7be8f4aed76d1caf423b3453e0824a7595","coverage":[{"denominator":32,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T10:54:06.184306Z","state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.16769/citation-record","integrity":"/paper/2501.16769/integrity","json":"/paper/2501.16769/citation-record.json","paper":"/paper/2501.16769"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2204.14198","last_updated":"2022-11-15T23:07:37Z","snapshot_observed_at":"2026-07-06T13:05:12.350238Z","submitted_at":"2022-04-29T16:29:01Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.14198","snapshot_observed_at":"2026-08-10T10:54:06.048139Z","title":"Flamingo: A visual language model for few- shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.048139Z"},"links":{"cited_paper":"/paper/2204.14198","citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:7ea21a6c782a08c476561d05ec5e33709bd99baa882c2cdb68d265fcbd61f403","observation_id":"6f91db66-a518-4fdf-860f-52a5e47d5eb1","resolution":{"observed_at":"2026-08-10T10:54:06.048139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.685015Z","title":"Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, and Amanda Askell et al","venue":null,"work_id":"aaa5c674-ce49-4b30-8f06-b565e6585b13","year":1901},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.053021Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:17d59949a0bd01cb39530df9c44e38a8417f9587e6c4ef9aebfa7298a29ebfde","observation_id":"75a1b2de-421c-4634-b4f1-63f5649ac2c3","resolution":{"observed_at":"2026-08-10T10:54:06.690980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.667685Z","title":"Zero-shot semantic segmentation","venue":null,"work_id":"7ba37715-14cc-49f4-bbea-b092c80ac8ca","year":2019},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.057178Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:fa456864731a64e33c4ad0f3f128c3a4a4f04e96432bbffac3754dc962ee43d2","observation_id":"14b24a7e-c4b8-45a2-8fe7-8882bff9b5dd","resolution":{"observed_at":"2026-08-10T10:54:06.674025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.650772Z","title":"Emerging properties in self-supervised vision transformers","venue":null,"work_id":"10d4fa79-6850-4d2f-950f-5fd9b7a01758","year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.061264Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:31d7c2fb8ad311d276aa4c9128bed5137b1be0867f1d439c52c22cc5ea3357f8","observation_id":"6592b3b3-f1d7-4c29-b710-2bf5831a6497","resolution":{"observed_at":"2026-08-10T10:54:06.655694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.634871Z","title":"An empirical study of training self-supervised vision transformers","venue":null,"work_id":"3e385a21-35e3-42fe-aa8e-60bce70a76a5","year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.065554Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:8d9cbe93c59949107fe692df8eff795b2b5f6c51466e580462409c1fc249066f","observation_id":"494d0fdf-2882-40a7-9ad2-78a5bd9a130f","resolution":{"observed_at":"2026-08-10T10:54:06.640062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.619934Z","title":"Bert: Pre-training of deep bidirectional transformers for language un- derstanding","venue":null,"work_id":"1cf80492-6b59-45ce-9b05-fdd8bb7fe4e3","year":2019},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.069805Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:267674f5b759c5f60627f6a83354966026e9a4567ac8f2cb8af5f79cb92a0b46","observation_id":"e0d3caa3-c80b-4217-bb43-ab3005c413db","resolution":{"observed_at":"2026-08-10T10:54:06.624670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.02311","last_updated":"2022-10-05T06:02:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-05T16:11:45Z","title":"PaLM: Scaling Language Modeling with Pathways","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.02311","snapshot_observed_at":"2026-08-10T10:54:06.074250Z","title":"Palm: Scaling language modeling with pathways","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.074250Z"},"links":{"cited_paper":"/paper/2204.02311","citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:2a9ad86f75e8e65ef91c220f4af868571525e5a2b081fd71351469a1e8564bfa","observation_id":"e064962a-193e-4628-a3a3-5882205b3903","resolution":{"observed_at":"2026-08-10T10:54:06.074250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-10T10:54:06.078400Z","title":"On the opportunities and risks of foundation models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.078400Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:f1a5d987e18c86de318cd5aeed6cdbeba1aca405c4c9382f72a6eed55d372435","observation_id":"aeab8e8e-f33f-4dba-a973-9bb8a0982768","resolution":{"observed_at":"2026-08-10T10:54:06.078400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.604699Z","title":"The pascal visual object classes (voc) challenge","venue":null,"work_id":"abda994d-b63e-4080-8d26-7883f9733f2f","year":2010},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.082565Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:92280c249963b56d26cd264499900b92619917a5855d5bd7160698fab1f11200","observation_id":"aea3d2e5-b556-4e1b-893e-29fce153770f","resolution":{"observed_at":"2026-08-10T10:54:06.609581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.589935Z","title":"Bootstrap your own latent-a new approach to self-supervised learning","venue":null,"work_id":"818a306c-60a4-4aeb-8064-602448907686","year":2020},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.086268Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:d15bb4f2d35ef34a65fb6d4b76545566f2d28be4632371d62b4bda86922b401e","observation_id":"1ff063a3-5c47-47a7-87c1-5bb2d5cce502","resolution":{"observed_at":"2026-08-10T10:54:06.595119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.574082Z","title":"Context-aware feature generation for zero-shot semantic segmentation","venue":null,"work_id":"0fae3272-cd57-4204-afb7-78151b2390d5","year":1921},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.090301Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:4d1240860696c60c109161fe4cd9c1e6305ca645e2ff185c067db07c9e4e56a8","observation_id":"5c880cc9-ef6c-45ab-9088-a5222b9a5247","resolution":{"observed_at":"2026-08-10T10:54:06.579730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.559594Z","title":"Simultaneous detection and segmentation","venue":null,"work_id":"a018a511-b737-451d-abac-0d4cf5551faa","year":2014},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.094096Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:e3e26cdc3e2a54fa80ab817fb31b16a2fcec05ac75b8adaf5dfb399788cb769d","observation_id":"4ee4c634-c2d1-45ab-9c05-fd954edc82c1","resolution":{"observed_at":"2026-08-10T10:54:06.563893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.14795","last_updated":"2022-03-15T22:37:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-30T17:53:34Z","title":"Perceiver IO: A General Architecture for Structured Inputs & Outputs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.14795","snapshot_observed_at":"2026-08-10T10:54:06.098569Z","title":"Perceiver io: A general architecture for structured inputs & outputs","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.098569Z"},"links":{"cited_paper":"/paper/2107.14795","citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:4bf55b2246fec848621f7c47c85be939429399a1e2b40f4b49a15b8707687f16","observation_id":"a61a390e-fbd0-4240-ab9c-1f6b8eecca94","resolution":{"observed_at":"2026-08-10T10:54:06.098569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.545186Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision","venue":null,"work_id":"7807c972-7d5b-4a85-8d18-274663f88719","year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.103351Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:088284b2f5da8b885b6c6ad0248ef358483f88c5ede19c5558baefd74d70f4e2","observation_id":"3bda8739-f5dc-4d22-bccd-2de83f9df1ad","resolution":{"observed_at":"2026-08-10T10:54:06.549229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.532035Z","title":"Deep spectral methods: A surprisingly strong baseline for unsupervised semantic segmentation and localization","venue":null,"work_id":"135a40e4-2978-4df6-9893-46d10bcfc8a0","year":2022},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.107703Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:0e167a822f449bf8d07e2d0e6c1b57c05317a3b6e7ba5a832fd6e5583d59a82e","observation_id":"52a6fcf3-2f09-4983-acdc-ff0452739398","resolution":{"observed_at":"2026-08-10T10:54:06.536186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2003.08934","last_updated":"2020-08-03T22:17:31Z","snapshot_observed_at":"2026-08-07T21:12:33.939201Z","submitted_at":"2020-03-19T17:57:23Z","title":"NeRF: Representing Scenes as Neural Radiance Fields for View Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.08934","snapshot_observed_at":"2026-08-10T10:54:06.112024Z","title":"Srinivasan, Matthew Tancik, Jonathan T","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.112024Z"},"links":{"cited_paper":"/paper/2003.08934","citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:f0418a704284be2815b88a85cb44759219cc1751ccbfc445d25ded6726347337","observation_id":"aeca5109-b1ce-4b1d-be25-f52da5c75bf6","resolution":{"observed_at":"2026-08-10T10:54:06.112024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.518552Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"14ab83db-1ce2-4dd8-a7e6-3ed008de774a","year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.116677Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:d912e637c7917fd18524665581b952ab88ab7d3c94c05150e676403d969e3efd","observation_id":"7bf68296-1297-4492-9a37-1627f9515d2e","resolution":{"observed_at":"2026-08-10T10:54:06.522822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.504386Z","title":"Conditional networks for few-shot semantic segmenta- tion","venue":null,"work_id":"5753173a-9cc4-42d1-9e3c-492e71a89684","year":2018},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.120860Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:9fc48fafe7a0aead875fba40441b9133903e3d7143942f5340a0c539411fcfca","observation_id":"62cbd38c-5f81-4833-ada5-b73883eda648","resolution":{"observed_at":"2026-08-10T10:54:06.508628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.490384Z","title":"Vision trans- formers for dense prediction","venue":null,"work_id":"988f4b18-f6ed-47a8-b415-651b40941dc9","year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.125173Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:c332d36a9cd9aa220f2cd0c84801d8dca375b3203457857be9d4a6da03655466","observation_id":"89ddb77d-8a7a-46de-a332-b0c8c9969170","resolution":{"observed_at":"2026-08-10T10:54:06.494736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.475119Z","title":"U-net: Convolu- tional networks for biomedical image segmentation","venue":null,"work_id":"53fbcd89-5d75-48b0-be4c-28c98e8a5057","year":2015},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.129392Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:83627be906b3c9f20da16348718e8c119616ed685caf6c54ac99e2dd30b9d665","observation_id":"9f59eb25-6a67-435b-8c08-e17b1739f039","resolution":{"observed_at":"2026-08-10T10:54:06.480252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.459910Z","title":"One-shot learning for semantic segmentation","venue":null,"work_id":"aed0452d-2cbf-4bc4-855d-9215a5e57235","year":2017},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.133810Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:c337c17332707e223511343a643a00f4f36e87897e9a6c08802e2bdde50eaa25","observation_id":"215dbaaf-f008-412d-9e1f-e8cc61a7ec10","resolution":{"observed_at":"2026-08-10T10:54:06.464797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.443700Z","title":"Fully convolutional networks for semantic segmentation","venue":null,"work_id":"f8c36311-644e-4ef3-8198-45f78fc3446c","year":2017},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.138184Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:622cfa29c3dec80e50ac4d0cf0f1a8d697373a495fe623f0e0d543bb14b5ef42","observation_id":"f59dcb86-ce78-4e8f-9a82-93f5a7a8aa4b","resolution":{"observed_at":"2026-08-10T10:54:06.449246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.428583Z","title":"Unsupervised salient object detection with spectral cluster voting","venue":null,"work_id":"98ee2b98-dd4f-459b-8d6a-7f77588d8f47","year":2022},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.142625Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:e0ac007c370b866e01f3f178efddb57e7d95d22d020e48aeda9d4c3a2138754b","observation_id":"1b89e5fb-d2f1-4467-8b86-95314d4991b9","resolution":{"observed_at":"2026-08-10T10:54:06.433134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.414311Z","title":"Oreshkin, and Martin Jagersand","venue":null,"work_id":"a60f4a1d-ee19-4b35-8ab3-930e488102ec","year":2019},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.147009Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:c900b872d35b20de694888d5ce8fcb4a4018e476384b853e0b433823e4126f95","observation_id":"713261a4-52b4-4cbb-bfc8-26da241b6462","resolution":{"observed_at":"2026-08-10T10:54:06.418859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.398950Z","title":"Implicit neural representations with periodic activation functions","venue":null,"work_id":"7b0968cc-8686-4f0c-900c-681b29701f10","year":2020},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.153490Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:4e2210d10f8b964d3145fb3d9580522fa106e159d8a014665257ccb23afb7a9f","observation_id":"30a4347c-7849-4d82-a040-d8d4fcc0dfab","resolution":{"observed_at":"2026-08-10T10:54:06.404206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.381879Z","title":"Fourier features let networks learn high frequency functions in low dimensional domains","venue":null,"work_id":"9380a0c9-4d4b-4d33-a69c-937fc1dfbc02","year":2020},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.157801Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:add76c2ec8455b4f555c04fe0378d97e995eb6e4ae4732d4cd54813b2f1910ae","observation_id":"92d4fcef-9cba-492e-a833-59694bc3adc8","resolution":{"observed_at":"2026-08-10T10:54:06.387498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.365857Z","title":"Menick, Serkan Cabi, S.M","venue":null,"work_id":"f4851fdc-08c3-4226-8ffd-42d98aa8e468","year":2021},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.162156Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:eb7300b1e2be4b9a6df6838a7316b68be7f6c47d1e605ee2cb466d12b801f9e3","observation_id":"dd9e2a18-daa9-4bda-af36-143e4919a601","resolution":{"observed_at":"2026-08-10T10:54:06.370664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.351531Z","title":"Attention is all you need","venue":null,"work_id":"641215f3-a225-4b94-a846-a1eef4d8eabe","year":2017},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.166384Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:5a460c11b649c978e1fa7f55eedd11555e364e6df4d6cf1181925fc3d728586e","observation_id":"2dff4929-f9d5-43f0-87b2-dfefc5905b92","resolution":{"observed_at":"2026-08-10T10:54:06.355776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.336606Z","title":"Panet: Few-shot image semantic segmentation with prototype alignment","venue":null,"work_id":"06f277f1-04a3-4588-9e8e-9d7e6078ea82","year":2019},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.171032Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:08c7314f49b20e1d4374941d1b6052bef7daaac9b720a7d9cb66cad70bfa5cd0","observation_id":"9e57a93a-c9bb-4a9f-baff-d7e282489966","resolution":{"observed_at":"2026-08-10T10:54:06.341091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.321864Z","title":"Semantic projection network for zero-and few-label semantic segmentation","venue":null,"work_id":"e4600555-9303-4e9f-8cf0-cbcd5d045801","year":2019},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.175446Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:196da734a171546a527bb7ced530a60abb254c801ef327d0de5255d31d3995a2","observation_id":"47b08435-b0c7-4d22-9af2-e8c8c5767cb5","resolution":{"observed_at":"2026-08-10T10:54:06.326957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.00598","last_updated":"2022-05-27T17:52:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-01T17:43:13Z","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.00598","snapshot_observed_at":"2026-08-10T10:54:06.179983Z","title":"Socratic models: Com- posing zero-shot multimodal reasoning with language","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.179983Z"},"links":{"cited_paper":"/paper/2204.00598","citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:1e6e7c9e50d9778d327f0f507ddc68a449839dc200be2d7ab5630f4058c49c41","observation_id":"2c1aaa83-0174-4e1f-885d-b82037429461","resolution":{"observed_at":"2026-08-10T10:54:06.179983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T10:54:06.305775Z","title":null,"venue":null,"work_id":"0f3b96bb-1894-4a0d-bc43-eb2f5da876c6","year":2020},"citing_paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T10:54:06.184306Z"},"links":{"citing_paper":"/paper/2501.16769"},"observation_digest":"sha256:31cbe7a109c86f6230700172b907662da90368f834c1833ba92effa4fff3af4c","observation_id":"078f95ff-dde8-4c1c-bb09-0c4d4628d449","resolution":{"observed_at":"2026-08-10T10:54:06.311218Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.16769","last_updated":"2025-07-02T01:46:17Z","latest_version":5,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T06:36:21.942137Z","submitted_at":"2025-01-28T07:49:52Z","title":"Beyond-Labels: Advancing Open-Vocabulary Segmentation With Vision-Language Models"},"reference_resolution":{"displayed":32,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":25},"total_outbound_references":32},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 32 of 32 outbound references and 0 inbound Pith citation observations for arXiv:2501.16769."}