{"as_of":"2026-08-19T17:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d076f2eda70c97581300e9ea891070b5de4e18594aae594b1d42727d1d329de7","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:42:01.350617Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.12194/citation-record","integrity":"/paper/2505.12194/integrity","json":"/paper/2505.12194/citation-record.json","paper":"/paper/2505.12194"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.193954Z","title":"Referit3d: Neural listeners for fine-grained 3d object identification in real-world scenes","venue":null,"work_id":"8cec085a-904c-4431-9b00-693877fa1ab0","year":2020},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.099614Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:9f3a88afd1708ee2f36b3d212acbfc5da125e6fd47e5e4197f105db88ac079c2","observation_id":"8b4be10b-1382-44bd-b4af-51394726bd5b","resolution":{"observed_at":"2026-08-15T20:42:02.199814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.104538Z","title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.104538Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:99cf8a4f188f5ce469a5c9eeeee0c4a7a3c1631019b8d28b0c69395640312a41","observation_id":"da35ada0-b447-4d57-b7e9-3dacc66b8a36","resolution":{"observed_at":"2026-08-15T20:42:01.104538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.165301Z","title":"Paligemma: A versatile 3b vlm for transfer, 2024","venue":null,"work_id":"0b438ad9-7ea7-4851-adf0-410cc6b80061","year":2024},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.108823Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:bb826f2aebbc90f6af1ea7e04a8a61fba405652067cfeee6ac8f4da71363d828","observation_id":"ed9d6ede-b858-48e3-b7f7-dbce13430eea","resolution":{"observed_at":"2026-08-15T20:42:02.171991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.151177Z","title":"Meyer, Yuning Chai, and Yong Jae Lee","venue":null,"work_id":"95265aaa-147d-4b4c-ae43-8cda5e2ff557","year":2024},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.113068Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:0516b0ae67c37a469e3b28c6ee3271480c6930c705558ee70a307bd6e85aae83","observation_id":"6acb85eb-f62b-4b90-b140-5ed25d4dc631","resolution":{"observed_at":"2026-08-15T20:42:02.155473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.12872","last_updated":"2020-05-28T17:37:23Z","snapshot_observed_at":"2026-08-14T22:50:00.733302Z","submitted_at":"2020-05-26T17:06:38Z","title":"End-to-End Object Detection with Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.12872","snapshot_observed_at":"2026-08-15T20:42:01.117530Z","title":"End-to-end object detec- tion with transformers","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.117530Z"},"links":{"cited_paper":"/paper/2005.12872","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:da4bf22c7f138e40549cfbcc4ebbb6101f11dcd5826bab32debfc44c0cfb3e18","observation_id":"6d3a2c34-fd32-4b0c-a84d-2f92adad4e50","resolution":{"observed_at":"2026-08-15T20:42:01.117530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.135506Z","title":"Chang, and Matthias Nießner","venue":null,"work_id":"46596094-993f-498a-b6f0-77fbbfad4b34","year":2020},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.122887Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:f9314d0af0782849f9018b9e02e0a431c2a92d0792fe123a82d7719a2d02c181","observation_id":"87d26fe7-bffa-416e-ab89-53b1cee56357","resolution":{"observed_at":"2026-08-15T20:42:02.140911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.117715Z","title":"Reproducible scaling laws for contrastive language-image learning","venue":null,"work_id":"baa22b68-afa5-4536-8204-d77a1577b0cb","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.127302Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:8286f3d4884214467ab532ccaf07db0d26d4cfaf11b3970a14a7e9962424217e","observation_id":"f83ae1f4-d6ec-4ce9-a251-543cbc481efd","resolution":{"observed_at":"2026-08-15T20:42:02.122862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.131352Z","title":"Gonzalez, Ion Stoica, and Eric P","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.131352Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:95b50bc962cf9645dc2c7c68d65eb7ac079caa9a133bbb0d615c3831f1cc91e5","observation_id":"39df7291-0c35-4e9b-8dad-793358ac81fd","resolution":{"observed_at":"2026-08-15T20:42:01.131352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.02311","last_updated":"2022-10-05T06:02:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-05T16:11:45Z","title":"PaLM: Scaling Language Modeling with Pathways","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.02311","snapshot_observed_at":"2026-08-15T20:42:01.135730Z","title":"Devlin, Jacob","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.135730Z"},"links":{"cited_paper":"/paper/2204.02311","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:b5c3bd0ea45382d00417445254dacc2e3ca33e387a10ee74492d0b389b8eba45","observation_id":"762ea977-ff9d-4b23-8948-1f2264653156","resolution":{"observed_at":"2026-08-15T20:42:01.135730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11416","last_updated":"2022-12-06T21:39:48Z","snapshot_observed_at":"2026-08-18T19:51:48.688199Z","submitted_at":"2022-10-20T16:58:32Z","title":"Scaling Instruction-Finetuned Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.11416","snapshot_observed_at":"2026-08-15T20:42:01.140159Z","title":"Scaling instruction-finetuned language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.140159Z"},"links":{"cited_paper":"/paper/2210.11416","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:56f3f6ad58125254034a356522a709407f88c61f81224ac2a1e5f9eaa60406db","observation_id":"f3c3f399-48b3-460c-b636-d75fea4626d4","resolution":{"observed_at":"2026-08-15T20:42:01.140159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.145285Z","title":"Instructblip: Towards general-purpose vision-language models with instruction tuning, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.145285Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:c097cf3154f56b4339b6fa3ea50f25a2a5b0d8dc72cd6441b11dbc8f69a23f85","observation_id":"f8130732-5a40-464b-a7ca-5f1466da59d3","resolution":{"observed_at":"2026-08-15T20:42:01.145285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.074170Z","title":"Qlora: Efficient finetuning of quantized llms, 2023","venue":null,"work_id":"a45a8c30-abb5-4fb5-9823-3e3c6d5a6c1e","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.149222Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:b27a3c1356fa07d0b502f4cd8dd35310289dbe84f6cd0da08c56bc924059587a","observation_id":"c45436c2-9cb1-4049-a484-6e16fa69a7d2","resolution":{"observed_at":"2026-08-15T20:42:02.079283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.057435Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding, 10 2018","venue":null,"work_id":"f81d4798-14fb-4def-91d5-afef3b8a7254","year":2018},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.153728Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:0b99b1783387f362d48655f5007474ca051b9a244a5b3771dca4fce2989c827c","observation_id":"37ccbc78-2b94-4772-9e69-0970e51f125a","resolution":{"observed_at":"2026-08-15T20:42:02.062339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.036215Z","title":"Eva: Exploring the limits of masked visual representation learning at scale, 12 2022","venue":null,"work_id":"1c785526-af21-420a-9269-a8c019e6294b","year":2022},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.157500Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:ec352ba383dd2d82c21e0d22932d511bf05d0aeb6f14a1cab7e0fc0168ca47b4","observation_id":"c9f7267a-0eec-42da-bda3-570e8bd66d9e","resolution":{"observed_at":"2026-08-15T20:42:02.042675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.018891Z","title":"Sebastian Borgeaud","venue":null,"work_id":"5e5ea841-0bd7-4a79-b129-bc22131c9949","year":2024},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.161837Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:31746c545412c198a9be94069a71c9b565adf827755ea68d636e77e6cd64e1f3","observation_id":"cf39f242-ab4c-42d8-a1da-c4bfbcc6a177","resolution":{"observed_at":"2026-08-15T20:42:02.024059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:02.002186Z","title":"Arzen-llm: Code-switched egyptian arabic-english translation and speech recognition using llms","venue":null,"work_id":"018dcbaa-c59a-4a29-bf59-de27b7931a41","year":2024},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.165947Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:4903bcda69b5575dcec6d691fbfbe1037ce73f324a605aba5cc23c1811241a1b","observation_id":"b58ed854-ed16-400c-9f2d-071448d86b51","resolution":{"observed_at":"2026-08-15T20:42:02.007589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.979562Z","title":"Lawrence Zitnick, and Ross Girshick","venue":null,"work_id":"4c10d4f7-9fd1-43a5-8cbb-415cbacca4db","year":2016},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.170276Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:9042ae370f2a69aeac68352717ac130e6399592957d007dd002d10b8b7b3d937","observation_id":"a5bc6b3b-bfbb-46e5-be73-e7cf9a4bf894","resolution":{"observed_at":"2026-08-15T20:42:01.984777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19785","last_updated":"2023-10-30T17:50:15Z","snapshot_observed_at":"2026-08-16T14:47:31.796733Z","submitted_at":"2023-10-30T17:50:15Z","title":"What's \"up\" with vision-language models? Investigating their struggle with spatial reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19785","snapshot_observed_at":"2026-08-15T20:42:01.174246Z","title":"What’s “up” with vision-language models? investigating their struggle with spatial reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.174246Z"},"links":{"cited_paper":"/paper/2310.19785","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:cedd1ce0cdd295570f07dd82b05ef88c299c3a11589de83ef589a172fe0d2c40","observation_id":"1e6b80dc-c27f-46a2-9f2c-b95a4af42833","resolution":{"observed_at":"2026-08-15T20:42:01.174246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.178917Z","title":"Referitgame: Referring to objects in photographs of natural scenes","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.178917Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:ee5fcd364501b00816432c2e8e7be2be3f48c3ce577ced5ecba8f96a9601bb73","observation_id":"cdb5dbec-3f04-4245-9e98-ac561733b61f","resolution":{"observed_at":"2026-08-15T20:42:01.178917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.950787Z","title":"Computational genera- tion of referring expressions: A survey","venue":null,"work_id":"f7ca81be-792d-4212-9456-0ba17ba7d212","year":2012},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.183124Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:858a14e4dfa58689336e7776c58dbe57eb743a5bdd1ddbed3910d4202db579ff","observation_id":"3927bc2f-8dbe-4b33-a61c-38541ce2b2d6","resolution":{"observed_at":"2026-08-15T20:42:01.956685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.935742Z","title":"When to retrieve: Teaching llms to utilize information retrieval effectively, 05 2024","venue":null,"work_id":"88f2329e-143c-4bad-b773-35c29bd7d62e","year":2024},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.187349Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:ff739035f6b9ae82d71d05b5aa553771421b74528696a6fb50a84bbd1e899a7d","observation_id":"50d03fe8-9096-468a-9ad8-5caeda702d2e","resolution":{"observed_at":"2026-08-15T20:42:01.940399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.920662Z","title":"Blip-2: Boot- strapping language-image pre-training with frozen image encoders and large language models, 2023","venue":null,"work_id":"d140e063-97f7-494d-b971-d245085494a4","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.194495Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:8153248b2c4e2ceabed5ab384807a4c15eb9697a492bbc804e75a831dfd466a2","observation_id":"5c85146a-578c-46bf-bcea-0d30960c5668","resolution":{"observed_at":"2026-08-15T20:42:01.926351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.201033Z","title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.201033Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:2bb8247469aabdfc2ac6ca4dcd038a88d6aacfc70226dbaee045511609ff8eb8","observation_id":"27718a0e-9a3d-4103-a71c-5aeb284cd0f7","resolution":{"observed_at":"2026-08-15T20:42:01.201033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.897526Z","title":"Selvaraju, Akhilesh Deepak Gotmare, Shafiq Joty, Caiming Xiong, and Steven Hoi","venue":null,"work_id":"516367f2-a214-4199-8ba9-3074356f3073","year":2021},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.205674Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:a983cebc2876045b158d7d650244fad7411f5e2dc7a7338e0ce2296257d57b81","observation_id":"5f3c152d-a8c4-4ad4-94e0-c62a001d1e4c","resolution":{"observed_at":"2026-08-15T20:42:01.902440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.879329Z","title":"Lawrence Zitnick, and Piotr Doll´ar","venue":null,"work_id":"659944c2-2ec1-4791-bc06-cb1511d1b95a","year":2015},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.210634Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:4edcf2b65a572432d0c628e7a4c24eab7e54410f2a1828330ffbec1e012f7fe6","observation_id":"d69a8043-cef0-40f3-93c5-14e17c9fe60a","resolution":{"observed_at":"2026-08-15T20:42:01.884073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.215353Z","title":"Visual spatial reasoning, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.215353Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:9294091a72c226309042b33905e86111d9242700dbb6c57722a2bc9928245bc7","observation_id":"ff3f4944-11d7-4a4e-9e26-782abbd2c978","resolution":{"observed_at":"2026-08-15T20:42:01.215353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.855368Z","title":"Refer-it-in-rgbd: A bottom-up approach for 3d visual grounding in rgbd images","venue":null,"work_id":"a7e47b38-6deb-4c38-879e-39d7baee3e48","year":2021},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.220484Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:d92f108fa6a1985c805a0b7070ce16ff8f56949c19b82659f212dd224deae53a","observation_id":"2e54f4f4-558c-4328-af62-5476b10f019b","resolution":{"observed_at":"2026-08-15T20:42:01.860306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.227573Z","title":"Improved baselines with visual instruction tuning, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.227573Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:5cae467dc6b506be1ac241a4fe1707740cfbd89ab3b13e1af94546babdab6516","observation_id":"378b87a3-6f1a-429d-8baf-4ab029211aa2","resolution":{"observed_at":"2026-08-15T20:42:01.227573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.232461Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.232461Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:c8b6b198fb805a2944ca022b02320034a5e13fd5225d17d21366d523ecb1d98f","observation_id":"3e2f3754-f5cc-4591-bcb2-dcc28e77cb6b","resolution":{"observed_at":"2026-08-15T20:42:01.232461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.815904Z","title":"Clevr- ref+: Diagnosing visual reasoning with referring expressions","venue":null,"work_id":"2497fcbe-c7ce-43e1-a3a4-76b92048b8a8","year":2019},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.237438Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:7fb5ce641a1fd7183baeede18719cc642d23ca2fcd37797d46b7efc5bff09b23","observation_id":"2ad51824-6f5f-4c72-9c8a-aec9b6d663c0","resolution":{"observed_at":"2026-08-15T20:42:01.822384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.800807Z","title":"Roberta: A robustly optimized bert pretraining approach, 2019","venue":null,"work_id":"8555f845-b92f-4501-a2f5-642cdd70ac10","year":2019},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.243529Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:1a6dd3c32dca8b82aa84c9fdcd635cc19c4f1d4838c4e7a0fbd9a940d12295cd","observation_id":"abaaab58-51d3-4651-a77c-354ec57b547f","resolution":{"observed_at":"2026-08-15T20:42:01.805674Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.06281","last_updated":"2024-08-20T03:56:03Z","snapshot_observed_at":"2026-08-17T03:48:18.599994Z","submitted_at":"2023-07-12T16:23:09Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.06281","snapshot_observed_at":"2026-08-15T20:42:01.248047Z","title":"Mmbench: Is your multi-modal model an all-around player? arXiv preprint arXiv:2307.06281, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.248047Z"},"links":{"cited_paper":"/paper/2307.06281","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:ecd424dcf89c7d60648c900c657039db1f9ac595f320fad769778d860b03998d","observation_id":"7891a45e-b672-4cea-9da9-f46cbd335944","resolution":{"observed_at":"2026-08-15T20:42:01.248047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.784171Z","title":"An embarrassingly simple approach for llm with strong asr capacity","venue":null,"work_id":"e225dcd2-5b83-4cfc-afc9-b7c0107fb200","year":null},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.254755Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:0d119abfcf8d7ab630bffa74dc4a7eab28c220060a121595f00c47cd33b40b49","observation_id":"f2dd3b37-32f9-4cbf-adc4-c732769c27d4","resolution":{"observed_at":"2026-08-15T20:42:01.789663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.769542Z","title":"Generation and comprehension of unambiguous object descriptions, 2016","venue":null,"work_id":"fef6612a-d4d4-4711-8aab-6961d89d5988","year":2016},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.259634Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:109a51f3f511ec7305dab572845cc18c9faea28346ec809785262a4ea455ef2e","observation_id":"14052c44-5cfb-4b4e-bfb2-aa3f316cb49a","resolution":{"observed_at":"2026-08-15T20:42:01.774052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.754514Z","title":"Sun-spot: An rgb-d dataset with spatial referring expressions","venue":null,"work_id":"42ff91eb-9ebd-4d8f-951f-94602c2eb250","year":2019},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.263919Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:0d54abc1e88c3d81f4c4195b85f78acde5fcf67b8db401704b80a5cae966e4d4","observation_id":"8f8cf1a7-e58c-4095-8fa7-3a664ea5b488","resolution":{"observed_at":"2026-08-15T20:42:01.759669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.738075Z","title":"Domain terminology integration into machine translation: Leveraging large language models, 2023","venue":null,"work_id":"0bc5750e-3330-4db2-a5af-4c8c91610f91","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.267895Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:3e475b59152d2af197e69fd307357f90f44d5c70aa25e61fc35271e2c8d2c212","observation_id":"6212bf1c-060d-4f35-8ddf-b9d89230f993","resolution":{"observed_at":"2026-08-15T20:42:01.744988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03152","last_updated":"2024-05-06T04:05:19Z","snapshot_observed_at":"2026-08-16T18:20:04.023557Z","submitted_at":"2024-05-06T04:05:19Z","title":"MMGER: Multi-modal and Multi-granularity Generative Error Correction with LLM for Joint Accent and Speech Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.03152","snapshot_observed_at":"2026-08-15T20:42:01.272339Z","title":"Mmger: Multi-modal and multi- granularity generative error correction with llm for joint accent and speech recognition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.272339Z"},"links":{"cited_paper":"/paper/2405.03152","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:e9902100b9133c72b6baa915c7bdcfce6e59331522e3c39b809733bc09d6513e","observation_id":"172f7d6a-b936-4f74-98e3-f8da32adf00f","resolution":{"observed_at":"2026-08-15T20:42:01.272339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T20:42:01.278138Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.278138Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:e13b159ff6af0ba1eb83d80513b3934be0f75a316cc99c7014058e3e8e0ac6e8","observation_id":"a287a7dd-1a97-4e49-97df-e94acd94d29e","resolution":{"observed_at":"2026-08-15T20:42:01.278138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.720287Z","title":"Reverie: Remote embodied visual referring expression in real indoor environments, 2020","venue":null,"work_id":"2db34ab5-26d9-46a3-a12a-e85ac2172677","year":2020},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.283316Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:f2462fa1344827f810d45684c442632fb914a426389ff7dc0ce14326aa8b7f30","observation_id":"9ef06a3e-801d-43e5-bbec-4cab472d661c","resolution":{"observed_at":"2026-08-15T20:42:01.725428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-15T20:42:01.288751Z","title":"Learn- ing transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.288751Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:7f544806dcca794edf84ecad7fd57633c37154425dc0c07dc1a1959fc59c292f","observation_id":"aa93edda-840a-48b5-a640-a9d3aa962d3b","resolution":{"observed_at":"2026-08-15T20:42:01.288751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.293525Z","title":"Sun rgb-d: A rgb-d scene understanding benchmark suite","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.293525Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:05cee83f96e11d59c021cca47f0f61f75464951f19abce560cf32346cca8c05f","observation_id":"5dac8927-2b91-471b-8687-8912fdb530b0","resolution":{"observed_at":"2026-08-15T20:42:01.293525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.686145Z","title":"Eva- clip: Improved training techniques for clip at scale, 03 2023","venue":null,"work_id":"ecb87a02-20f0-43dc-8d29-a3de7d694685","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.298020Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:e466662dc39c975dd5e98852d6cd0e7fa7caa73fac74b5ffd06d5c3f3a3bef62","observation_id":"1e05419f-41f1-41e1-b87b-2e32960c3ae2","resolution":{"observed_at":"2026-08-15T20:42:01.693187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.668974Z","title":"Self-retrieval: Building an information retrieval system with one large language model","venue":null,"work_id":"779d1436-4763-4028-b74b-75919dae666e","year":null},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.303168Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:091447606a1ad1bc297ae745f267db58c7403f2b93a748f1b636c70afbe3a298","observation_id":"ba500ec0-b31a-4592-9f9c-68fd8781acca","resolution":{"observed_at":"2026-08-15T20:42:01.674285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-15T20:42:01.311814Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.311814Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:7d6997c4fc1ffe50e5351c1555006b3707086b9aa0d64952ae448801d9e0d63b","observation_id":"7cc07cf8-ec25-4113-a282-d956bbc8a967","resolution":{"observed_at":"2026-08-15T20:42:01.311814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.649417Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"8372484c-f4b7-47b6-a4d2-126b9e0f5f72","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.318526Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:bd7cf05bc5b92e2c6b8c98802d3a860bd6011900cbce6d370cf7a551437cdb5f","observation_id":"f5505bb0-e851-4ef3-a301-8ecbf0ea6c15","resolution":{"observed_at":"2026-08-15T20:42:01.654685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.632526Z","title":"The generation of natural descrip- tions: corpus-based investigations of referring expressions in visual domains","venue":null,"work_id":"3a5c9036-5e01-4ce5-8c64-04d9701116c9","year":2022},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.322797Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:dcaa6b070fa1746e63083a449d0fc97e7e74b6e6be1e50bb31aa0a21c14e3331","observation_id":"3f3ad556-9c77-41fb-b97d-86ec82214b33","resolution":{"observed_at":"2026-08-15T20:42:01.638951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.327938Z","title":"Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.327938Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:87baa605ee263ee46bdef08344662c58569784fa3ffd4179ff4a0ec89300baa3","observation_id":"90b8c13a-315c-4806-982e-dacf13fa1c13","resolution":{"observed_at":"2026-08-15T20:42:01.327938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.605253Z","title":"Xlnet: Generalized autoregressive pretraining for language understanding, 2019","venue":null,"work_id":"75ef63e9-f72a-4e05-81e9-a426cc3dc335","year":2019},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.331978Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:b4790f8ee6b1922f3574ce36f65de977145582f62a448c3743a6045fb199829c","observation_id":"2d377aad-26fb-44de-89cb-11dafc29afbe","resolution":{"observed_at":"2026-08-15T20:42:01.609997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.590002Z","title":"Modeling context in referring expressions","venue":null,"work_id":"a3e15b22-faeb-4e42-af8b-96463f59d7ff","year":2016},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.336192Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:6c8ca88249616ceed4b2c0f5606835b73a78e261506dbb226de91042dc076c1e","observation_id":"200d4cde-550a-45b7-a4d3-9273d800c70f","resolution":{"observed_at":"2026-08-15T20:42:01.595606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.573945Z","title":"A joint speaker-listener-reinforcer model for referring expressions","venue":null,"work_id":"44c2e43b-bf66-48d3-8b43-d932f35dae10","year":2017},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.340627Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:212fe7dca239955ce3b7b7c9289cff601f27431e783e263183f4b367c4fd245d","observation_id":"0361d342-2114-4f40-aa67-82e7f06e30b6","resolution":{"observed_at":"2026-08-15T20:42:01.579028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.558897Z","title":"Prompting large language model for machine translation: A case study, 2023","venue":null,"work_id":"3c0bce65-f622-4759-9bcb-c923684a5639","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.345926Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:6a36ec539435574ef83bea56b73ef47ea947cce9a230e5ef1de843713e0ddafb","observation_id":"6ceb79ce-3570-4b12-aa65-c996ba87e7e2","resolution":{"observed_at":"2026-08-15T20:42:01.563240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:42:01.543117Z","title":"Toolqa: A dataset for llm question answering with external tools","venue":null,"work_id":"1d25248f-ab68-4382-9163-5a5f969b4568","year":2023},"citing_paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T20:42:01.350617Z"},"links":{"citing_paper":"/paper/2505.12194"},"observation_digest":"sha256:ab8d3374ba474dcfb410b8eb672f7b99263ab79fe45169d2a64197b1f91b9489","observation_id":"d2ff4c5a-c55a-4eb0-bf61-eb5f79fce442","resolution":{"observed_at":"2026-08-15T20:42:01.549350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.12194","last_updated":"2025-05-18T02:07:55Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-19T07:01:34.918945Z","submitted_at":"2025-05-18T02:07:55Z","title":"Spatial-LLaVA: Enhancing Large Language Models with Spatial Referring Expressions for Visual Understanding"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":19,"verified_exact":0,"verified_fuzzy":33},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 0 inbound Pith citation observations for arXiv:2505.12194."}