{"as_of":"2026-08-08T08:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5eb5c0ff5ee1bfdfd33b733d2968ea5f2bda2c5e97de609de2269cb902a74288","coverage":[{"denominator":65,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T01:55:35.172193Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.05916/citation-record","integrity":"/paper/2606.05916/integrity","json":"/paper/2606.05916/citation-record.json","paper":"/paper/2606.05916"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:92586ef821802a38d6f4adf23710d0eaab4af84b0d46968aad19e547e38d7a1f","observation_id":"41d3524c-66c7-4e15-9e27-7758dcb75232","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:fa74e67a1357eac285c0fb7f288d727c08eefc4750ece15590edb664d223914b","observation_id":"fb20e177-68f7-49c9-a149-a8d1dfa32661","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"End-to-end object detection with transformers,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:54506f5ea674f058460f1ce5d434ec3660c01cc8a1513912389342ca969d8bca","observation_id":"316b1f5a-8039-4173-ae0c-2dd7a823e948","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Localized vision-language match- ing for open-vocabulary object detection,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:b72ac80f6a56271fe13368ae953cf2edd7b6173885354bcb0e4d43d604544716","observation_id":"ebffe1f4-2c86-4d8c-89d5-e73259876bc5","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Detecting twenty-thousand classes using image-level supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:164cd25d50dc45b9ebbc43531e6a11ecea9ed64935680970bf4cab351ca3f8dd","observation_id":"8ed61ff6-5fd7-4574-b089-cb3227312fef","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14843","last_updated":"2022-11-27T14:47:31Z","snapshot_observed_at":"2026-08-06T03:25:36.980355Z","submitted_at":"2022-11-27T14:47:31Z","title":"Learning Object-Language Alignments for Open-Vocabulary Object Detection","version":1},"cited_work":{"arxiv_id":"2211.14843","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2211.14843","snapshot_observed_at":"2026-07-02T12:46:56.511625Z","title":"Learning object-language alignments for open-vocabulary object detection","venue":null,"work_id":"16206e2c-d31b-4a44-975c-a6859ead74a9","year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/2211.14843","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:dc132fd290716a21f6ddbe714fe64453aa07c9d9ef4622a5d52c489bc380b9ba","observation_id":"0436a92f-5dc4-4ea1-bed1-1f9ce3e30e3e","resolution":{"observed_at":"2026-07-02T12:46:56.513029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Ex- ploring region-word alignment in built-in detector for open-vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:bd51f0bef4090fb2347cde86351faf4cbbcd3517e07b468640f495aa6c4407fb","observation_id":"c795d932-3cc3-4df8-a9d6-a4d4de5a5e60","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Contrastive feature masking open-vocabulary vision transformer,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:d0a12fa3f22d106a981aa33689c8c265805cbad3b09b47f8ebc23b996f2f5243","observation_id":"efd16259-e96b-4a5b-8b9a-322547787724","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.13921","last_updated":"2022-05-12T01:27:40Z","snapshot_observed_at":"2026-07-06T11:04:30.929441Z","submitted_at":"2021-04-28T17:58:57Z","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","version":3},"cited_work":{"arxiv_id":"2104.13921","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.13921","snapshot_observed_at":"2026-07-04T16:29:57.549240Z","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","venue":"cs.CV","work_id":"59541f32-18cf-4328-ad60-c9018e1401cf","year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/2104.13921","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:c6273c7a74892be056620b1cc2589470b19cc407241c74519eaba4e1a83f5ed1","observation_id":"d158545e-6da1-47df-b6b1-3e4989b700e7","resolution":{"observed_at":"2026-07-02T12:46:56.500741Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Aligning bag of regions for open-vocabulary object detection,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:9c1fe74991f5311c03e7d6460e7dba7665838e08df472ff9b0bfaf0dc42d3e73","observation_id":"3d009faf-c6e5-49fb-a960-0b1531040ae8","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Learning background prompts to discover implicit knowledge for open vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:7bc10128cec62753e338809457b4425ba87a9730f4ff9339761eaf8b98cf842c","observation_id":"762c0fcb-ddf9-49c7-a4e1-b4020369ced6","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Conceptnet 5.5: An open multilingual graph of general knowledge,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:098ce7964d316bf2bf43125828f7a69016d9a89a8cc9028370705dd2a9c2027a","observation_id":"e133d708-d430-4429-a63a-6e4857202b00","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Visual relationship detection with language priors,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:9070a88578ce7cac60356ccf2a62fc68e83ac256adb478620ac940be67094bcb","observation_id":"6ee7fa75-cccd-42ce-b8e2-43feed1f468c","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"You only look once: Unified, real-time object detection,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:1cae674a17c3e8bdff010b08616eb55299a2c04b241de3adb6b98610059764dd","observation_id":"8dd3e332-ecec-4e20-9d01-a3c678ec729d","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Mask r-cnn,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:7a6c8fe65c60d16df077aa8f39d77eefb2f8553d3a558750bc4e4b55fe9458ff","observation_id":"4d5ad88c-f596-44c4-bacf-8f8caaa0576f","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Sparse r-cnn: End-to-end object detection with learnable proposals,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:106d610d1e7803d566703a9d3c484bb451f20311d3f3a823cb2c8f7506686fec","observation_id":"8e3d3f5c-37a6-4bca-abc1-32aceb579bde","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Zero- shot object detection,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:b74c8d5487fecf05e0ebea887abd8999cc7220847d9a124159a7239487a9e3c0","observation_id":"7936a253-8b83-4f12-ab95-6948cf43c7d9","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1805.06157","last_updated":"2018-05-17T07:59:37Z","snapshot_observed_at":"2026-08-07T01:23:49.701365Z","submitted_at":"2018-05-16T07:01:27Z","title":"Zero-Shot Object Detection by Hybrid Region Embedding","version":2},"cited_work":{"arxiv_id":"1805.06157","doi":null,"metadata_source":"pith","pith_arxiv_id":"1805.06157","snapshot_observed_at":"2026-07-02T12:46:56.507010Z","title":"Zero-Shot Object Detection by Hybrid Region Embedding","venue":"cs.CV","work_id":"e0ca2a8c-e309-4df9-a7c1-87ee47c04606","year":2018},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/1805.06157","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:e84ea3f50267c4a8f8565b954680a8dafd5bb3b229318d8e6ff836273cf37403","observation_id":"55d2df89-bbda-40dc-8333-5859dfb14808","resolution":{"observed_at":"2026-07-02T12:46:56.508128Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Synthesizing the unseen for zero-shot object detection,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:c952ff320416124967732e0ca647fff9db0b725de27ba06702345c16e62f2eae","observation_id":"7e625f8d-e42f-4dce-8666-60fb1262b63e","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Transductive learning for zero- shot object detection,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:461fb5f7e2e45b07ca6710d052044bd64f9f5d62cbe3c02df4852559934d54e9","observation_id":"f00fcf6f-bcb7-4baa-8ac9-028557d79beb","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Don’t even look once: Synthesizing features for zero-shot detection,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:4374530a5822f45cc2625883c4eb286fe875d8a1a617d3b223cec75c61b3dca0","observation_id":"984006fc-30c0-458d-8b3a-fff4f194d810","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Gtnet: Gen- erative transfer network for zero-shot object detection,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:cc3f81a8aa1fe0255fb4acfbb953ca1fe1a89496fab8cb3653d96cef263b407e","observation_id":"c99db987-16f3-4612-b8f8-90488d4ab143","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Object-aware distillation pyramid for open-vocabulary object detection,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:527bbb2bab7fa1285ebbecc41ca7489e1b53c35ce21a4c0c8177f9f1fe19e1ff","observation_id":"04fec66e-bc14-47ac-8fe4-5a99f413a011","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:2560b52399ab5973cd85af420fdbeede01bc13ccd6701d7ab5bccc56ce51b3ea","observation_id":"dcfbc152-1efc-4413-a15a-676377f2f2a6","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Scaling up visual and vision-language representation learning with noisy text supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:76176c5a1c01cfd07766e3753e3fb71d0c85d84e83219388e196c00b0212d58e","observation_id":"63c6a44c-2a96-42ae-b451-4c9d984fc8b3","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Learning visual features from large weakly supervised data,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:2ac6ff83c8709598c65fd10f2a292207c7c06d9ba19f5a83cc2b3393a2dc232e","observation_id":"170733f8-8784-434a-9495-5c86e0a8c0e1","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Learning visual n-grams from web data,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:f9cafc63915678cf50a7bc9878c2c229bc4a2044e0ff3aaea0c00a8caa758433","observation_id":"32c1650d-426d-47c9-8dc3-df8f7f390933","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Imagenet: A large-scale hierarchical image database,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:37401016dcedd7179ebbfc1085d60253cde9a8bdd42fb4cfb88f3fc580bcd77e","observation_id":"fcad276e-320d-425a-93e6-2aaa5a872592","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Groupvit: Semantic segmentation emerges from text supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:3bb6a2d1d5eb3005f9e7ac1fdb7ef20a0b3fac90fb0a4f19b6edcce3dc8df90b","observation_id":"40c3f33f-6399-4888-939f-02f4106995d3","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.03546","last_updated":"2022-04-03T03:33:43Z","snapshot_observed_at":"2026-08-01T06:14:31.660540Z","submitted_at":"2022-01-10T18:59:10Z","title":"Language-driven Semantic Segmentation","version":2},"cited_work":{"arxiv_id":"2201.03546","doi":null,"metadata_source":"pith","pith_arxiv_id":"2201.03546","snapshot_observed_at":"2026-07-07T21:34:09.314172Z","title":"Language-driven Semantic Segmentation","venue":"cs.CV","work_id":"20f0fe18-203a-461e-b860-9b93636dd77b","year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/2201.03546","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:8004acbaf002e542f469279539e1367c6b57203533de7a9062e92308ab907aa1","observation_id":"b929a7f0-f3bd-4daa-9f80-803f3b7e1a55","resolution":{"observed_at":"2026-07-02T12:46:56.510530Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Extract free dense labels from clip,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:17eb0a7e14c38adcab3d3e2cf2895d60f8fb54fb8597c04f9e0651c2f41ccda0","observation_id":"1935832e-f386-462b-bad4-54a6a4f963d0","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Scene graph generation by iterative message passing,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:494b61535edcd661b5b773b385591ca97d883718f1e7c4a874ee6cb2ebdbb698","observation_id":"bf13f684-4bc8-414b-866f-7c703ba0adf7","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Integrating object-aware and interaction-aware knowledge for weakly supervised scene graph generation,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:a02753024caa68cb52021327a2259ddcfe0e4a15a13b3acd589c2da639210545","observation_id":"e559496d-a64d-467a-8fc7-70cd39e95584","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Bipartite graph network with adaptive message passing for unbiased scene graph generation,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:430618fc413d06eaf5f7960e126d1e21c783f5e6eaffc19e1354fde91642c25b","observation_id":"227d365f-3da0-49b2-9b3f-c4e109dd2e7c","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Sgtr: End-to-end scene graph generation with transformer,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:2d4008b5506d02e195a6573980dc16080fdaf9524335b3de19d82c6271cfd779","observation_id":"0f08f59d-798a-44f3-ad93-8ca95124eb4e","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"vtgraphnet: Learning weakly-supervised scene graph for complex visual grounding,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:cd2a4c6d82f97ff0bdb0968070552b741cf9fa0ab7fb67703c7de1e8ce2240ed","observation_id":"7a003565-0612-4338-a6b6-fc61a7454968","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Structure inference net: Object detection using scene-level context and instance-level relationships,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:bb028a958b9e581bfe4546a272136c14cd1608751af81da3f587490c94daf781","observation_id":"d7c82740-7e0e-4535-82cb-3dd88afbac17","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Graph-structured referring expression reasoning in the wild,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:1aef95a9acb105f622ad3953baf29759c656740fe2d94f06605de7d01eb946d2","observation_id":"e45e1bd6-3702-40c4-8b38-c697af8617ea","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Towards open-vocabulary scene graph generation with prompt-based finetuning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:86d5fce23acb6f5433cd8b097b737313cfdf87f9478d6ae7c29f5d30e68adf12","observation_id":"9dd033e6-8fe4-43cf-8bcb-42e7e99cecc8","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Learning to generate scene graph from natural language supervision. in 2021 ieee,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:be96aeb32691a97d16f5f821ad00e2755d13a81121e7430336fcf08570964165","observation_id":"7989a3b2-63ea-4df8-aa13-7c651f59e250","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Unbiased scene graph generation from biased training,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:5037982c6840850470c5bb76a6c1b9a31ceabc5ad47161829bfe6ef9b6190443","observation_id":"2f76a8b3-5005-4718-acb6-f7e1f83412a3","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:e4f19595b83818a9c19568554400e3c661e9992f42c143d3dd0d39ec4937dd49","observation_id":"b1a14419-9927-4839-bcc8-27e784ecf4af","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.08415","last_updated":"2023-06-06T01:53:32Z","snapshot_observed_at":"2026-07-06T05:01:27.910364Z","submitted_at":"2016-06-27T19:20:40Z","title":"Gaussian Error Linear Units (GELUs)","version":5},"cited_work":{"arxiv_id":"1606.08415","doi":"10.18653/v1/n19-1122","metadata_source":"pith","pith_arxiv_id":"1606.08415","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gaussian Error Linear Units (GELUs)","venue":"cs.LG","work_id":"0466fd22-03a1-4a61-af0a-a900e77bb023","year":2016},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/1606.08415","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:db535ac35f4e5159e7b6b4599034c241e8fc0524f641f42705ef5deeb388d0a3","observation_id":"56023be6-162c-4ce8-bc32-7f8d288a18ec","resolution":{"observed_at":"2026-07-02T12:46:56.511290Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Generating semantically precise scene graphs from textual descriptions for improved image retrieval,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:5af2a43b2b31a6d477e16140e3736816a8005aa15acd642f1b6d8b9da77b093e","observation_id":"2d127189-23c8-444e-a95a-32fdc3084d99","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Open-vocabulary detr with conditional matching,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:37c01e26f2c7a9035325c69c16c2841a8f93276a88a8c7d6e93182dfa1a5e02a","observation_id":"9b09a38d-fb94-4589-9e4f-f1e781d164a7","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Promptdet: Towards open-vocabulary detection using uncurated images,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:1155419eab39babc7192cc8626f42287fe2f36530907a95a76f6ee860e7cb22f","observation_id":"324e52aa-0340-47e5-b9e3-410e5e5b0050","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Dis- tilling detr with visual-linguistic knowledge for open-vocabulary object detection,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:cd6e5d7fa45cdea38f1da23c12962cdc301ae02b89f7d6c5266dac1d00821df4","observation_id":"5b870b3f-7385-40dc-a19c-5e5a48d7eb1c","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Open-vocabulary object detection using captions,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:5294051f79f5ffcc6a1b519ee7268114cc3353e22dc69aa81830fca53376628c","observation_id":"c6265e2b-ef7f-458b-9665-b7bf410143ba","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Proxydet: Synthe- sizing proxy novel classes via classwise mixup for open-vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:8903ca21bfb6dfb9b9ffd7d68c7e95bcff794ba86fd4950519329442889c7c33","observation_id":"aebeab44-0c25-42fe-94f6-a707a90f2bf7","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Retrieval-augmented open- vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:6904f9d56061281d3baf377e6a4a1a9b2f7421b8c5299f36f50e6d0fd9366b43","observation_id":"ed568baa-967a-4402-b3e6-958797676fa1","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Benefit from seen: Enhancing open- vocabulary object detection by bridging visual and textual co-occurrence knowledge,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:24ffdc60c314f894900812ce8d2fdf4068c4d7e4a3a43b85b5b835eb6a045349","observation_id":"ceb9d2b6-a77f-4e73-972c-d87a10984c03","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Regionclip: Region-based language-image pretraining,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:33bb55a189855d175c3d7e28d679010af57535c15b62fc0772bc3bc27ea29044","observation_id":"7b2476b2-5ee6-41ce-81a1-f0681a9a0d64","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Learning to prompt for open-vocabulary object detection with vision-language model,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:f3ed19964a8d77bdd9d723e92f2e752218b6761f78c2ed30c43b49fd83d3b947","observation_id":"0629e89d-392d-40c0-ae6f-53cfd56df91a","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Exploring multi-modal con- textual knowledge for open-vocabulary object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:2a0bec5b8c19cceceddff4b3b8a2052b84425a5067af38094d8d09d24b8500dd","observation_id":"3961ea6f-8c64-4934-a80a-320d9dcc73ba","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.15639","last_updated":"2023-02-23T19:14:52Z","snapshot_observed_at":"2026-07-06T13:58:22.032846Z","submitted_at":"2022-09-30T17:59:52Z","title":"F-VLM: Open-Vocabulary Object Detection upon Frozen Vision and Language Models","version":2},"cited_work":{"arxiv_id":"2209.15639","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2209.15639","snapshot_observed_at":"2026-07-03T00:57:30.385197Z","title":"F-vlm: Open-vocabulary object detection upon frozen vision and language models","venue":null,"work_id":"2e2ab981-0561-40ab-a804-85d58df32d35","year":2022},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/2209.15639","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:47e48fc3816c1d75fd3f282aa05722655ed04d44f28916ee4e85a78a650ca48f","observation_id":"9024ba1f-423c-43e9-b086-b71fa430e75e","resolution":{"observed_at":"2026-07-02T12:46:56.506522Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Microsoft coco: Common objects in context,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:e53dd18fb452c80a73bab8fa59e90e027ba86ed2aff1d971198da5684d6bab01","observation_id":"c52254ff-4fed-4412-aa9c-996e2d9750f9","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Lvis: A dataset for large vocabulary instance segmentation,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:6c23c332a0225f20c4ef6b41ea8873eb1982f2a9ddbcf684b07000f78e4b1c36","observation_id":"44558679-f173-459e-8b08-3cfe15ecd8db","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1504.00325","last_updated":"2015-04-03T20:21:16Z","snapshot_observed_at":"2026-08-04T18:05:27.145522Z","submitted_at":"2015-04-01T18:13:43Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","version":2},"cited_work":{"arxiv_id":"1504.00325","doi":"10.48550/arxiv.1504.00325","metadata_source":"pith","pith_arxiv_id":"1504.00325","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","venue":"cs.CV","work_id":"b3d6fb46-4169-4a28-8f7e-2ca6774211da","year":2015},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/1504.00325","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:350a1dab822fcf6af300fe1bc69fe333339690e786111ce40bbd56d666ba65ba","observation_id":"5bd28bc8-817e-4b0d-a91b-bf9a8d040f9d","resolution":{"observed_at":"2026-07-02T12:46:56.515418Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image cap- tioning,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:92572d35f8f391f59379feafe3679e1f48c6e2023e71339a61078308fce7de7a","observation_id":"4d67b278-10cf-4169-9b45-b30bfda507b6","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Pytorch: An imperative style, high-performance deep learning library,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:bb6fc7dfe7179474dad4628857247314a666e1795c6261f13e1a7bfc79a47875","observation_id":"d80e8228-4610-4f7b-867f-7fb0cc5893a0","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":"2010.11929","doi":"10.1175/jcli-d-22-0357.1","metadata_source":"pith","pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","venue":"cs.CV","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","year":2020},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:a1d11866364a03ce3d2a7524b864c0a7bb147a8489b8a9edb371fbab09616c51","observation_id":"12f100ba-b0b4-4759-9042-eb41e0c23924","resolution":{"observed_at":"2026-07-02T12:46:56.498120Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Feature pyramid networks for object detection,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:25e4c62897b378afba98f2f1dd942cca4f8d69f5876e82bb4d6f77c21a3a29b4","observation_id":"55b740f8-0b7f-497b-8ff2-0686f6413d96","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Aligning pretraining for detection via object-level contrastive learning,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:37c93f21fa63be9bf9c741d2cfd42ea0a40af90a479080125fd9e225dcba3af2","observation_id":"c9783fab-8b2b-40e0-9d47-e9756db45bb8","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Objects365: A large-scale, high-quality dataset for object detection,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:5ac8fc1ff8e74c0919e9892c9d8e65062301d897af589809f0a8c310f891bf9e","observation_id":"379d76d8-f9ba-48cf-9876-45aa88b2927e","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:55:35.172193Z","title":"Grad-cam++: Generalized gradient-based visual explanations for deep convolutional networks,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-28T01:55:35.172193Z"},"links":{"citing_paper":"/paper/2606.05916"},"observation_digest":"sha256:4441416fd6060096c405b464f3741ed557fc6759eb9b46476639b7e53a30646a","observation_id":"e27d7cf4-d5fe-4759-ab9c-a00732da2240","resolution":{"observed_at":"2026-06-28T01:55:35.172193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.05916","last_updated":"2026-06-04T09:23:08Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T17:05:26.813562Z","submitted_at":"2026-06-04T09:23:08Z","title":"Unveiling the Unknown: Open Vocabulary Object Detection with Scene Graphs"},"reference_resolution":{"displayed":65,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":57,"verified_exact":8,"verified_fuzzy":0},"total_outbound_references":65},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 65 of 65 outbound references and 0 inbound Pith citation observations for arXiv:2606.05916."}