{"as_of":"2026-08-10T08:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5fe4144e629ef336fe8b04e6746929089fef4f73ff8baa507a68d2813cb0eeb0","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:57:07.508659Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.19891/citation-record","integrity":"/paper/2507.19891/integrity","json":"/paper/2507.19891/citation-record.json","paper":"/paper/2507.19891"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.233663Z","title":"Generic attention- model explainability for interpreting bi-modal and encoder- decoder transformers","venue":null,"work_id":"3e0cba49-2b8c-432e-99de-35264bc66fd7","year":2021},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.341932Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:58f7def046ff9257573d4596b0cadc92344a2c45739a4bfaa2b360bcd424d572","observation_id":"a1320da5-9f30-4a53-be01-ab63d838a8d6","resolution":{"observed_at":"2026-08-06T13:57:08.238290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.215735Z","title":"Transformer inter- pretability beyond attention visualization","venue":null,"work_id":"3aef5357-ee18-407b-82b4-41d5c519ceb5","year":2021},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.346927Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:1479fae07fd81b9252637750a5b3514861f078d5e7ed0d87fe3914a22641e26f","observation_id":"4f2ab381-69fe-4024-90e8-eab32803ce81","resolution":{"observed_at":"2026-08-06T13:57:08.220968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.198394Z","title":"OvarNet: Towards open- vocabulary object attribute recognition","venue":null,"work_id":"5ec0fe3f-820f-403d-ac4b-e74ce37fdc8a","year":2023},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.351230Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:d1dd3b0d4c0451b06f5ec128da728a42cfec38ff116784017d79b54a4ec36edb","observation_id":"b2e1122d-0e44-40ab-b0cb-7094af1c3a1f","resolution":{"observed_at":"2026-08-06T13:57:08.203160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.177398Z","title":"Re- verse attention for salient object detection","venue":null,"work_id":"0add54e8-8b3d-45e4-90cb-a4180b253a65","year":2018},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.355783Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:662491e750957acbde42d06f3c3a9ff34b58638aa5d3a580cacd3cf213c1f740","observation_id":"9b03aa13-67a8-46aa-b325-088fc1ee4f8c","resolution":{"observed_at":"2026-08-06T13:57:08.182162Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.159071Z","title":"Spatial- rgpt: Grounded spatial reasoning in vision-language models","venue":null,"work_id":"88ac8815-aace-42db-83b0-fe83700438c1","year":2025},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.360611Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:82e4dd2efd2a7766ce0555b6d14ef38d8b16e258a96bda8e9e8dae4a705fdb87","observation_id":"e7f73bc2-e61d-4d59-aff2-22a9f5135098","resolution":{"observed_at":"2026-08-06T13:57:08.164389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.144097Z","title":"The Pascal Visual Ob- ject Classes (VOC) challenge","venue":null,"work_id":"7af22006-d78b-4769-b67e-6a4f43c95dd2","year":2010},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.365062Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:03156e76f61d0473c8f56abf7c9d4bed91d6c07e130ab25bf9c6070574418fad","observation_id":"8c83e2d8-b370-4277-8135-a3d15a16dbea","resolution":{"observed_at":"2026-08-06T13:57:08.148906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.128684Z","title":"Open-vocabulary object detection via vision and language knowledge distillation","venue":null,"work_id":"0a24353a-169a-4bd5-889f-306e188d3fac","year":2022},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.369891Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:9e0b013b75dd9eb06f7348f4a0813394461bd13c5a30dca0d2b277f5b80fe705","observation_id":"2fd21171-2887-484e-8811-ecc412e3fc40","resolution":{"observed_at":"2026-08-06T13:57:08.133226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.110723Z","title":"Attention-based multimodal fusion for video descrip- tion","venue":null,"work_id":"fedc0c34-83f7-43ba-acee-c19b9b989af3","year":2017},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.373883Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:17b85ed149f3ae089527a645699ba19b3bf8aa327bcb8676e846beab459a04b2","observation_id":"2434f22c-64f5-4a9f-84ae-c8914739a1cf","resolution":{"observed_at":"2026-08-06T13:57:08.117503Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.092936Z","title":"Semantic segmentation with reverse attention","venue":null,"work_id":"0800ca2e-3470-4147-b647-cd1df6f5ee53","year":2017},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.378794Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:7d5f26f49faf504a909a5904916b807ff9f7cb6ab02c4820eeee535772bcf000","observation_id":"3aa6e1bc-e160-4600-bcf3-fdb0159d4b71","resolution":{"observed_at":"2026-08-06T13:57:08.097489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.079062Z","title":"Scratching visual transformer’s back with uniform attention","venue":null,"work_id":"b28579eb-5730-43f3-a066-818897626893","year":2023},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.383727Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:d5c8b5464a7bada56212e8473d73ffa8f27cb431eef40185ebb1fb1ce1bebce6","observation_id":"76f18aea-4c91-42a3-8e35-1ea232a3e6e2","resolution":{"observed_at":"2026-08-06T13:57:08.083545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.060947Z","title":"Attention is not expla- nation","venue":null,"work_id":"9a22af0d-d5fd-4df7-a1f8-50f00efa91cf","year":2019},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.387755Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:2c70090896d926088021d8211d59a9bd800e9b96b71d9e8e006f49e0285dc3e9","observation_id":"1f75648b-c629-497a-88a8-f030251d5628","resolution":{"observed_at":"2026-08-06T13:57:08.066680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.046581Z","title":"MDETR- modulated detection for end-to-end multi-modal understand- ing","venue":null,"work_id":"f2f54a0f-d2f0-4e1a-a633-2dbd2618f2e7","year":2021},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.392358Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:08b179c1c7207208ea89b61f79f702088b408654d038ef3be3ea75aae8a8c854","observation_id":"295c4892-c186-48c4-98c5-003afce6a47a","resolution":{"observed_at":"2026-08-06T13:57:08.050848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.031799Z","title":"Shallow and reverse attention network for colon polyp segmentation","venue":null,"work_id":"d5bc7dc1-ced9-40c9-b15e-cb05a3c10fdb","year":2023},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.397579Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:a14a660bf4b0b71095e87e9f5c9b893d45e4095c3f0f11b3a297e7b85fc7a5e7","observation_id":"3fbc8ff4-c7c8-4254-a0f3-ced10887080f","resolution":{"observed_at":"2026-08-06T13:57:08.036471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.402219Z","title":"Grounded language-image pre-training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.402219Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:5c5b4d1d96eeae9b195fa0ab9a3cfe374752c5006a382aa3b9b34da5e2281bdd","observation_id":"05df848d-5c7f-49da-b543-baa3492aa3e5","resolution":{"observed_at":"2026-08-06T13:57:07.402219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:08.001844Z","title":"Distilled re- verse attention network for open-world compositional zero- shot learning","venue":null,"work_id":"f6dfd3e1-0869-48ec-bf5f-d473e57dd798","year":2023},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.408049Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:e60e605aaeeed54070140dc1067510c96ed80f28c2bc29bac92f788e4025fca0","observation_id":"a685fe3f-90bb-4c75-84d9-cd92430c9757","resolution":{"observed_at":"2026-08-06T13:57:08.007228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.984797Z","title":"Rta-former: Reverse transformer attention for polyp segmen- tation","venue":null,"work_id":"0bccf460-276d-4868-8817-08bfec6c6eff","year":2024},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.413006Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:52f74c8cc68f57398ecb2df59f9c41468c4441ffcb59c66707fc22044387f58b","observation_id":"bddf920d-580f-4ab3-9add-2326e29000fa","resolution":{"observed_at":"2026-08-06T13:57:07.990406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.970417Z","title":"Attention is not enough: Mitigating the distribution discrepancy in asynchronous multimodal sequence fusion","venue":null,"work_id":"4f81c662-7837-42c2-8fbd-faca6d7b93a3","year":2021},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.423632Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:4b1f109f90a30c83ed1d9ac8744647e29e4fb27349d405e62f669537aac3279b","observation_id":"c751a0b3-9110-4651-93d3-b332510f5b60","resolution":{"observed_at":"2026-08-06T13:57:07.974681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.956879Z","title":"Faster R-CNN: Towards real-time object detection with re- gion proposal networks","venue":null,"work_id":"ccc9e18e-7838-42b0-a6b3-452033ae6e6e","year":2015},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.427686Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:ec1ee137d3948ac2fd365bb1be544499f90d904c2ae5db4107099d12a3b371be","observation_id":"d84431c8-e50f-4970-b872-7236a78b0f49","resolution":{"observed_at":"2026-08-06T13:57:07.961130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.939485Z","title":"Self- attention with relative position representations","venue":null,"work_id":"36deb879-4205-485e-8278-42846e720394","year":2018},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.432250Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:fe26576548c3ffae068d3b3c07905db119fbb02f89138d8b6fddef2aad84dfe9","observation_id":"a4bfaaec-1406-47ba-8107-06424423b834","resolution":{"observed_at":"2026-08-06T13:57:07.946824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.923150Z","title":"GroundVLP: Harnessing zero-shot visual grounding from vision-language pre-training and open-vocabulary ob- ject detection","venue":null,"work_id":"b930d5fc-7357-402a-aa08-99577cab941a","year":2024},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.437179Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:72b8ef95d9d096d9d7f3fd07e68af22a3d602107aa7a436f4418eb577f8ab8e7","observation_id":"f1b5e668-fb99-4194-96f0-1d18f3439088","resolution":{"observed_at":"2026-08-06T13:57:07.928386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.907804Z","title":"Zero-shot open-vocabulary OOD object detection and grounding using vision language mod- els","venue":null,"work_id":"476f44a8-0590-4208-af3c-1a67f5d1d2dd","year":2025},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.441500Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:6fb133010a43484817c3e15d999c754233005c764ef4e7eeed5731f5f97f769e","observation_id":"55c22222-f5d6-4650-9e67-41ca9d74e052","resolution":{"observed_at":"2026-08-06T13:57:07.913182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.884726Z","title":"Reverse and boundary attention net- work for road segmentation","venue":null,"work_id":"8412af29-9eeb-444d-aecc-4ce8781242eb","year":2019},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.445925Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:c4eb350c3f35c6ec8ab738947fb5f1b8404ef5c86cfed2948164025f58b37970","observation_id":"2744a0dd-e5a2-4ee8-8c7e-193ccae3ef1e","resolution":{"observed_at":"2026-08-06T13:57:07.892700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11976","last_updated":"2025-06-22T00:39:23Z","snapshot_observed_at":"2026-08-09T16:06:21.482138Z","submitted_at":"2025-06-13T17:34:05Z","title":"How Visual Representations Map to Language Feature Space in Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11976","snapshot_observed_at":"2026-08-06T13:57:07.449816Z","title":"How visual representations map to language feature space in multimodal llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.449816Z"},"links":{"cited_paper":"/paper/2506.11976","citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:4a38c355456e170a11d2b5ce886fcae1fbc200740988cd8d2514e5e9cdc5bab5","observation_id":"7f81683f-e654-4363-ad77-03ebc2dffce9","resolution":{"observed_at":"2026-08-06T13:57:07.449816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.867765Z","title":"OV-VG: A benchmark for open-vocabulary visual grounding","venue":null,"work_id":"6c0a79ac-1eb2-4762-929e-aa1b4e053507","year":2024},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.453984Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:be7cbdd044ddef7dd417049819da36e878cbd059baeba53187b989c554c8b23f","observation_id":"0f9841ac-1976-41b8-ae7f-6794fe6e462c","resolution":{"observed_at":"2026-08-06T13:57:07.872157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.849938Z","title":"Ra-net: reverse attention for generalizing residual learning","venue":null,"work_id":"bb2e3d29-a451-4dc8-8e69-3bf57b8add6e","year":2024},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.457869Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:b3070c1f1ef48e15d919073974ef104c23dae804351783a23532a440076d4616","observation_id":"d61aeec8-e96c-43f7-9172-a93b4fdd5c96","resolution":{"observed_at":"2026-08-06T13:57:07.856092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.832965Z","title":"Attention is not not expla- nation","venue":null,"work_id":"13d31f85-78ad-4432-b901-74a12753bcec","year":2019},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.462446Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:1e1521cef609a495708dbacbca052b51bdbd5d6b602c631aef4229560f94e4cd","observation_id":"e1d15086-c7cb-4e11-bfb7-5fd367987be6","resolution":{"observed_at":"2026-08-06T13:57:07.837660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05901","last_updated":"2025-01-13T02:34:19Z","snapshot_observed_at":"2026-08-08T08:03:26.182607Z","submitted_at":"2025-01-10T11:53:46Z","title":"Valley2: Exploring Multimodal Models with Scalable Vision-Language Design","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05901","snapshot_observed_at":"2026-08-06T13:57:07.467028Z","title":"Valley2: Exploring multimodal models with scalable vision- language design","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.467028Z"},"links":{"cited_paper":"/paper/2501.05901","citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:d18a61570705ecd79a2438e5b9a732ac132c7f652a41ed1cd3e5a8efb2f6cf6e","observation_id":"baa7f74c-5f0a-411f-8950-4950baa3dbff","resolution":{"observed_at":"2026-08-06T13:57:07.467028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.810801Z","title":"Im- age inpainting with learnable bidirectional attention maps","venue":null,"work_id":"70dc6f86-2985-4ee4-aafc-e39907588096","year":2019},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.471294Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:9995995c8b9af46a94a5ad4b51877b0fa9562d9794a3936919585efef9e5430e","observation_id":"14ee3606-6dd7-468d-aacd-a944dde597c7","resolution":{"observed_at":"2026-08-06T13:57:07.816496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07905","last_updated":"2025-06-09T16:20:54Z","snapshot_observed_at":"2026-08-07T21:24:53.935343Z","submitted_at":"2025-06-09T16:20:54Z","title":"WeThink: Toward General-purpose Vision-Language Reasoning via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07905","snapshot_observed_at":"2026-08-06T13:57:07.475851Z","title":"WeThink: To- ward general-purpose vision-language reasoning via rein- forcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.475851Z"},"links":{"cited_paper":"/paper/2506.07905","citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:179dc63c331309db0eba0b1e0703809f172610bc260dd3a76322a6965614f14f","observation_id":"db245629-0bcf-49c4-9b7c-5aa648952409","resolution":{"observed_at":"2026-08-06T13:57:07.475851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.792535Z","title":"RLAIF-V: Open-source AI feedback leads to su- per GPT-4V trustworthiness","venue":null,"work_id":"74bb9529-5f31-4c9d-b7f0-2cf5b266eb95","year":null},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.480531Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:06f4d3e8deb353c556152ae630c1f1584b7dbda97c0959ca0bf5a2bf297b4c4f","observation_id":"2b038f6b-b78f-4c5a-80e1-c68c6af65714","resolution":{"observed_at":"2026-08-06T13:57:07.798576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.776885Z","title":"Open-vocabulary object detection using captions","venue":null,"work_id":"d5848f99-b56d-4db0-8346-606cc77f3d4a","year":2021},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.484958Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:c554b1fbf55f4c190107d1f7c639fb6135453dbd4984aff5ec2d0604f7cddd0e","observation_id":"88239d37-135a-4996-97ba-896e2e29db6f","resolution":{"observed_at":"2026-08-06T13:57:07.781027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.489518Z","title":"LED: LLM enhanced open-vocabulary object detection without human curated data generation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.489518Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:0aeabd796b459ebe8dcf77039212eafe728477d0d161b2b7674d37cfbcba05e0","observation_id":"53864601-f6ff-4093-af85-523a13130653","resolution":{"observed_at":"2026-08-06T13:57:07.489518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.762413Z","title":null,"venue":null,"work_id":"64da2609-5a4e-4ce6-a98b-ea7764b64c77","year":null},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.494507Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:a5d394815f4845ee908cf48ff53dd7a1a2bdeeedb52f4fc74c19f6d3c4aa59ef","observation_id":"1692c408-b1fe-4478-ae5b-11bdee89b0ab","resolution":{"observed_at":"2026-08-06T13:57:07.766952Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.740708Z","title":"Thus, we examine the correlation between the area of ground truths (normalized to image size) and those of de- tection Abox","venue":null,"work_id":"9dc62ff8-e455-45fa-9645-751417250f4b","year":null},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.499540Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:976f642467e190399c32506de6842904a868a302f2b6d25809a865cf13205bec","observation_id":"c48ce24a-1dea-4a2f-8b21-cdf43ace8d7f","resolution":{"observed_at":"2026-08-06T13:57:07.752196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.724999Z","title":null,"venue":null,"work_id":"b32e008c-b4a7-4235-b3f7-fd46553afd17","year":null},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.504234Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:801de27e44c0ea9cf3894159b3ea42db426199a077ce295561b5eaea5a0b2126","observation_id":"37d86476-9cf6-49bd-81b2-58af6624cd51","resolution":{"observed_at":"2026-08-06T13:57:07.729723Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:57:07.706479Z","title":"In the first case of inverse-distance reweighting: α′ ij = 1 1 + γ|αij − m| , which peaks at αij = m and decreases as αij deviates from m","venue":null,"work_id":"a351546d-59a4-46f1-bdbb-f29c0d5b3653","year":null},"citing_paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:07.508659Z"},"links":{"citing_paper":"/paper/2507.19891"},"observation_digest":"sha256:6539fa531f5a71b417d68a0424fc7d1dab882f260c4ab8b7ddb5996dc19d8e2f","observation_id":"f64ca025-9c90-4b78-9244-1a14d7714f80","resolution":{"observed_at":"2026-08-06T13:57:07.713577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.19891","last_updated":"2025-07-30T04:47:07Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T03:50:13.576578Z","submitted_at":"2025-07-26T09:43:09Z","title":"Interpretable Open-Vocabulary Referring Object Detection with Reverse Contrast Attention"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":29},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 0 inbound Pith citation observations for arXiv:2507.19891."}