{"as_of":"2026-08-08T02:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7a0b1282494ff43e1a337d5aacc922464025b495131d88cd7e9692e250264140","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-24T10:24:52.653023Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2212.12130/citation-record","integrity":"/paper/2212.12130/integrity","json":"/paper/2212.12130/citation-record.json","paper":"/paper/2212.12130"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zero-shot object detection","venue":null,"work_id":"64542cd2-1297-402d-a101-81f11cb0c90a","year":2018},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:4a88fa761fea96dfcbbc4c3f095fc5a8b2020000be1a44ed939f549d634bfea7","observation_id":"1853fddf-c2dc-438f-ba4d-d8d7d0cd85fc","resolution":{"observed_at":"2026-05-24T10:26:08.799353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cascade r-cnn: Delv- ing into high quality object detection","venue":null,"work_id":"79456c00-5f11-46d7-b2bf-0218a96dd5c3","year":2018},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:8c532de05b2fbde4203bc6a9af410351c6fedd8d6355e359b27ce982e513b1a2","observation_id":"14aa6167-a86e-463e-bbb5-aff63010defc","resolution":{"observed_at":"2026-05-24T10:26:08.802383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.12872","last_updated":"2020-05-28T17:37:23Z","snapshot_observed_at":"2026-08-05T04:05:42.353930Z","submitted_at":"2020-05-26T17:06:38Z","title":"End-to-End Object Detection with Transformers","version":3},"cited_work":{"arxiv_id":"2005.12872","doi":"10.48550/arxiv.2005.12872","metadata_source":"arxiv_reference","pith_arxiv_id":"2005.12872","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"End- to-end object detection with transformers","venue":null,"work_id":"f65d79c7-1ade-4a0e-b71e-140e377dbfb4","year":2005},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"cited_paper":"/paper/2005.12872","citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:12e07e15276af7459728bf33c4e10ba0903a73caabb5ac02b10b38187aafd5f9","observation_id":"33cb2f8f-3063-40e0-9f42-1e64c7f1b2ed","resolution":{"observed_at":"2026-05-24T10:26:08.075288Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1504.00325","last_updated":"2015-04-03T20:21:16Z","snapshot_observed_at":"2026-08-04T18:05:27.145522Z","submitted_at":"2015-04-01T18:13:43Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","version":2},"cited_work":{"arxiv_id":"1504.00325","doi":"10.48550/arxiv.1504.00325","metadata_source":"pith","pith_arxiv_id":"1504.00325","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","venue":"cs.CV","work_id":"b3d6fb46-4169-4a28-8f7e-2ca6774211da","year":2015},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"cited_paper":"/paper/1504.00325","citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:6e4f124224cfe348b0537cd4a2debc940ed3907de621b691d83ae05e2d67ea8e","observation_id":"dcaf4239-7be8-442b-998d-32601fd7840e","resolution":{"observed_at":"2026-05-24T10:26:08.085742Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dynamic convolution: Attention over convolution kernels","venue":null,"work_id":"a0f3e370-5788-42e4-823d-81767bf69119","year":2020},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:98efa1b6d94abb18c604cd9ecb5267c2bcf03361aa4fe7c6a096b74997827ca5","observation_id":"ef140cb6-d9e9-44d3-bbf5-66a2a5b2a811","resolution":{"observed_at":"2026-05-24T10:26:08.850934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deformable convolutional networks","venue":null,"work_id":"3da6161d-146c-4ef2-acb8-90f65c990c1a","year":2017},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:9fd30e852e5bde8a48e034eb3dd2596b83c12df3205c1fc9a093b29a226191b1","observation_id":"34f44890-ba2b-4533-b912-cb0bc2d25263","resolution":{"observed_at":"2026-05-24T10:26:08.859445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to prompt for open-vocabulary ob- ject detection with vision-language model","venue":null,"work_id":"41690271-5d7f-416b-9aa5-21e70fd8745c","year":2022},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:8dc20fec40b7ba43d455cfc262653e45db2a7668ee1000e8c9167e7594e752d9","observation_id":"d8136274-0dfc-4309-9fca-b1fdbf9a8d75","resolution":{"observed_at":"2026-05-24T10:26:08.862179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The pascal visual object classes (voc) challenge","venue":null,"work_id":"4e635ce4-f9a8-4525-83cc-58c7b8fbf270","year":2010},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:083e7460e191312bea9218be9ebe13a920342a6ed8367a3f9e900dacc13d12a5","observation_id":"fbaa615d-2915-493e-8df4-bed2274b6619","resolution":{"observed_at":"2026-05-24T10:26:08.843681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open vocabulary object detection with pseudo bounding-box labels","venue":null,"work_id":"c7e0f85f-839e-4ac4-beaf-b0bdcd4e77f4","year":null},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:9df6ec7523b7c3ad1e61b517eb3aecfcd34d632ce6d7d0044cdc6fd68708e1c0","observation_id":"bf433ac8-26ab-4e34-b68c-828f5b76809c","resolution":{"observed_at":"2026-05-24T10:26:08.853977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fast r-cnn","venue":null,"work_id":"f8b1f98c-f758-4c93-9b55-e04294854032","year":null},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:1f2fd126fda4165ebd4505178461be6f1f445fea075682fbff87470524fd7e39","observation_id":"e86bdcc8-6995-4ecc-981e-fce3d2e328c5","resolution":{"observed_at":"2026-05-24T10:26:08.817568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.13921","last_updated":"2022-05-12T01:27:40Z","snapshot_observed_at":"2026-07-06T11:04:30.929441Z","submitted_at":"2021-04-28T17:58:57Z","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","version":3},"cited_work":{"arxiv_id":"2104.13921","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.13921","snapshot_observed_at":"2026-07-04T16:29:57.549240Z","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","venue":"cs.CV","work_id":"59541f32-18cf-4328-ad60-c9018e1401cf","year":2021},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"cited_paper":"/paper/2104.13921","citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:e8de14e558172678226db33753cb37f5a7b8b6bfcca68df010203e96229da4a3","observation_id":"ca05504c-2f6e-4dea-b66f-39d5f3e9b40b","resolution":{"observed_at":"2026-05-24T10:26:08.092865Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LVIS: A dataset for large vocabulary instance segmentation","venue":null,"work_id":"981afec2-81aa-4e40-b183-c6f495f078a3","year":null},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:6729ad0d289ed80236176d3327cb045f3eecda884f982d4cd71301e407b422d1","observation_id":"871bf1b2-0f0f-42ec-9594-e48982ae8496","resolution":{"observed_at":"2026-05-24T10:26:08.821630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to segment every thing","venue":null,"work_id":"89591c15-b323-4a68-8ca2-d1c0e06bb1ef","year":2018},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:88dddfbb97e5d0cdc4601d98ae685dd7721273f5841b4dab67422b71214bdb08","observation_id":"e50e80d7-ab14-4a91-94fb-8117cdd305bc","resolution":{"observed_at":"2026-05-24T10:26:08.856511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open-vocabulary instance segmentation via ro- bust cross-modal pseudo-labeling","venue":null,"work_id":"dce03861-7cdb-45e1-a730-7fbaee582eae","year":2022},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:4596ba323ef6331fb08d8302d1280960ccc31771f254e4fd4204a65d9fafafdd","observation_id":"92d336d7-0ce5-418f-a352-72baaa3f8548","resolution":{"observed_at":"2026-05-24T10:26:08.821353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Spatial transformer networks","venue":null,"work_id":"51e5256d-fee4-4ad8-8f4e-31ba3364ce5f","year":2015},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:81af83521b69e7d5f56cfc1634f95e0c714152e4f338e9971bb006adf85a50c4","observation_id":"54634b0f-591b-4e89-97bd-91f54c42cf0f","resolution":{"observed_at":"2026-05-24T10:26:08.833425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":"f40d90a4-7b16-4c7a-ad1c-9ff74545cba4","year":null},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:a6a1ff3fdfd4e7e6111cc3c1698b3823e178187008444b98a974695007d3e43a","observation_id":"29de3d94-ef53-450a-bdf4-120f3b89c30e","resolution":{"observed_at":"2026-05-24T10:26:08.884181Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dynamic filter networks","venue":null,"work_id":"6d34963b-0f0c-4c2e-959c-cb307bacbaa8","year":2016},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:5e7b0058783f05faf1290e77fb0dfeadd5557e87b46547602362c8a6d6086813","observation_id":"03cacb02-225e-4b43-9590-38e8411fcf2e","resolution":{"observed_at":"2026-05-24T10:26:08.805033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Shapemask: Learning to segment novel objects by refining shape priors","venue":null,"work_id":"3a8ef3d8-35a9-44d3-be7f-8e5db72cfa08","year":2019},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:0ec7620210310c9f4a93cddbb00bb99eb0befc46be19025a148eb7067ae4af16","observation_id":"7aa0c7bc-a9c3-4458-99a3-d03c9e82bd4f","resolution":{"observed_at":"2026-05-24T10:26:08.889646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"729bd281-1706-4be4-8739-b834167b3f77","year":2014},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:e6b3f7fdfeea63095662aeee2009ea8a51e2a9156cf3be53607da7d0406985f1","observation_id":"40a964e5-16cf-4152-9135-20e8efce85bd","resolution":{"observed_at":"2026-05-24T10:26:08.891830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ssd: Single shot multibox detector","venue":null,"work_id":"750e777e-2a19-4479-9e22-4d885d01126f","year":2016},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:43b8558b06b2c518638ec5e562c91d0476dcb1a0a34906ee6dd83d42c9ba1d3e","observation_id":"35468cb8-b49a-45ba-8508-ecba344e5eae","resolution":{"observed_at":"2026-05-24T10:26:08.876780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.06230","last_updated":"2022-07-20T12:24:25Z","snapshot_observed_at":"2026-07-06T13:09:25.356940Z","submitted_at":"2022-05-12T17:20:36Z","title":"Simple Open-Vocabulary Object Detection with Vision Transformers","version":2},"cited_work":{"arxiv_id":"2205.06230","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2205.06230","snapshot_observed_at":"2026-07-04T12:59:53.224246Z","title":"Simple open-vocabulary object detection with vision transformers","venue":null,"work_id":"4a029835-bd63-4b98-ab43-d2ff7887490d","year":2022},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"cited_paper":"/paper/2205.06230","citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:44cf7409721cb158e516c7fa852955b4c472177cfda274127a972467287bdacf","observation_id":"4390f54e-8a0f-49fa-b181-681292ccbac0","resolution":{"observed_at":"2026-05-24T10:26:08.081039Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learn- ing transferable visual models from natural language super- vision","venue":null,"work_id":"b5587c3f-7031-4d33-9c3e-6533043294ca","year":2021},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:189b65ab10605658caeac05df85b1a6bc04043c23ebe2eed7da5eb045a87c37d","observation_id":"1f5eba6c-d079-4d21-94e6-bf9543c16c83","resolution":{"observed_at":"2026-05-24T10:26:08.868175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improved visual-semantic alignment for zero-shot object detection","venue":null,"work_id":"583f8d50-e854-4e7b-8859-69079f0a1f86","year":2020},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:3c52d1e0a29da74878c8a42c3902232c5e64321d84362182dd9f11e5c23d7ab4","observation_id":"e575f38e-8fba-4e96-af92-0ffc5294c366","resolution":{"observed_at":"2026-05-24T10:26:08.874242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"You only look once: Unified, real-time object de- tection","venue":null,"work_id":"f46efd17-6137-4adc-a46d-d9734c10353a","year":2016},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:5b402e3ef71ab25a9d151503726d57ac6a3aa03136cec0163581b74d5d71f804","observation_id":"625567e8-4aca-4b9e-b880-895459589fb9","resolution":{"observed_at":"2026-05-24T10:26:08.879498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Yolo9000: better, faster, stronger","venue":null,"work_id":"4cafa651-62fa-4b4c-b0bf-cbda0d27383e","year":2017},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:d7237f8518d37c88f9114226a0999ecbd4bbefc54f1e6ee832a07e7a13d7b05a","observation_id":"eb3318ca-651a-47db-9446-5eed3e04eb7e","resolution":{"observed_at":"2026-05-24T10:26:08.871525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks","venue":null,"work_id":"ca79fba5-21b1-44e4-84ba-bd337259cb83","year":2015},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:3f8d65e00a38f78b17fb821a7b8ea03acf3650d58330c74d9a3b9b337fc10b96","observation_id":"e9f88f37-c863-4213-ae0e-04e6fa39fe5e","resolution":{"observed_at":"2026-05-24T10:26:08.887190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Objects365: A large-scale, high-quality dataset for object detection","venue":null,"work_id":"55157587-4eea-4657-9b2d-637720711211","year":2019},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:e3a4d2e5e053adcbb88792e51e26f955be70c26782b1cdee0f605564324f999e","observation_id":"53eb6fb1-8032-486f-a48c-9da08b68cec2","resolution":{"observed_at":"2026-05-24T10:26:08.894724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Conceptual captions: A cleaned, hypernymed, im- age alt-text dataset for automatic image captioning","venue":null,"work_id":"d6ac37c7-859a-4228-994a-838d2cd922b2","year":2018},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:828e4d272439aa6c2f776cc08c20bf2295025497b2e1d0721688e48b6689045b","observation_id":"9b95e5b2-a46a-45bc-9539-2cbc3c88355b","resolution":{"observed_at":"2026-05-24T10:26:08.865028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Condconv: Conditionally parameterized convolu- tions for efficient inference","venue":null,"work_id":"17f6e242-7732-4962-8b60-0fadc378da04","year":2019},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:460e4fce1850bed0972390946d5dbd1cd562af20d5fddebb959da691d2e58bdb","observation_id":"a9d239b8-c980-49aa-8a39-a0b4d073b39f","resolution":{"observed_at":"2026-05-24T10:26:08.853685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11876","last_updated":"2022-11-30T02:42:54Z","snapshot_observed_at":"2026-08-06T18:03:15.880678Z","submitted_at":"2022-03-22T16:54:52Z","title":"Open-Vocabulary DETR with Conditional Matching","version":2},"cited_work":{"arxiv_id":"2203.11876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.11876","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open-vocabulary detr with conditional matching","venue":null,"work_id":"e00b7d9c-9ffe-4ced-90f8-f939165722fb","year":2022},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"cited_paper":"/paper/2203.11876","citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:e00536f9cc60ce8268dc22c2b3f524c750b761bd39bddba239bcb1366c2bc995","observation_id":"adca4fe5-0c32-47ad-8294-a734990014b1","resolution":{"observed_at":"2026-05-24T10:26:08.097198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open-vocabulary object detection using captions","venue":null,"work_id":"e7ebdb51-e86c-48e2-a3fb-a10705c6181a","year":2021},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:fbe50a22967c39a0f01b7e0d56fce4860ad65a6b3307ea7d29255ac6c2474666","observation_id":"3ff8ba1e-ed20-4eed-9048-87586d237015","resolution":{"observed_at":"2026-05-24T10:26:08.829842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Regionclip: Region- based language-image pretraining","venue":null,"work_id":"40d4803b-0c9b-4568-8b63-4bf688a545db","year":2022},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:354a89bbc684fdf0e0ed45c98c8086cf826746d2b841c04a6043eaf95f8820bc","observation_id":"bc497430-6a99-4d46-a075-7264a44cb7b3","resolution":{"observed_at":"2026-05-24T10:26:08.836669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to prompt for vision-language models","venue":null,"work_id":"21ed208c-fa9e-44ff-818c-4b33095d207a","year":null},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:3b60dfdaeab1044927311b067899016ad57c95887ee6cc3770cf128aa08ad76e","observation_id":"ff3ea736-9201-4243-b8e7-55a4e8c67120","resolution":{"observed_at":"2026-05-24T10:26:08.844473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Don’t even look once: Synthesizing features for zero-shot detection","venue":null,"work_id":"e6440ecf-0988-4f4f-bc19-aca01c1ea92c","year":2020},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:17a47dde3bab56f37624331d102a2dfe611058ab5c4849b7ab3bf099562047af","observation_id":"663676d5-8151-40e0-80ea-e64b074b43e9","resolution":{"observed_at":"2026-05-24T10:26:08.826113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"De- formable convnets v2: More deformable, better results","venue":null,"work_id":"5e55a803-7baf-4751-9b79-100cb1f27a06","year":2019},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:f49442a1230ccbed29b6acfb6d41b3e85d8b0059dff0391d08295010154d63ed","observation_id":"5ba1451c-913a-4fd4-8997-b08104201c12","resolution":{"observed_at":"2026-05-24T10:26:08.847956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deformable detr: Deformable transformers for end-to-end object detection","venue":null,"work_id":"7f361a76-725f-488f-a4cf-d502957d8acd","year":2020},"citing_paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection","version":7},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-24T10:24:52.653023Z"},"links":{"citing_paper":"/paper/2212.12130"},"observation_digest":"sha256:63fb8f145951edb5f6f4101f997fb0e88c1561c07c79068a5cee44d6f10b4a5f","observation_id":"54ff1a0a-da72-4667-8f6b-962d713ec7aa","resolution":{"observed_at":"2026-05-24T10:26:08.840626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2212.12130","last_updated":"2026-05-15T03:31:02Z","latest_version":7,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T14:34:19.510288Z","submitted_at":"2022-12-23T03:54:59Z","title":"Learning to Detect and Segment for Open Vocabulary Object Detection"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":5,"verified_fuzzy":31},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 0 inbound Pith citation observations for arXiv:2212.12130."}