{"as_of":"2026-08-08T21:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:02219e91aada9ed73e7cbfbb138a444b4007f06f88cf0437d0848543f2ab602b","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T07:02:05.336238Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.24810/citation-record","integrity":"/paper/2607.24810/integrity","json":"/paper/2607.24810/citation-record.json","paper":"/paper/2607.24810"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:57.534474Z","title":"Creating xBD: A dataset for assessing building damage from satellite imagery,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:57.534474Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:f358fab5314038c974c6d29feb2c91d281bdbb9debba409b23c72b6f8f242715","observation_id":"021478be-d6e0-436f-85ed-2d8eec7ccc64","resolution":{"observed_at":"2026-08-02T07:01:57.534474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:57.631021Z","title":"A biologist’s guide to the galaxy: Leveraging artificial intelligence and very high-resolution satellite imagery to monitor marine mammals from space,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:57.631021Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:a6802ae4e6a44cfce99dcc8111d38c779205da3cedbc4be33f5eab0ae2ee92bf","observation_id":"5cde64dc-df05-4f4b-8e6a-b8024fcf9c30","resolution":{"observed_at":"2026-08-02T07:01:57.631021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:57.727050Z","title":"Military image captioning for low-altitude UA V or UGV perspectives,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:57.727050Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:f4529a9f87f9dd330d340e827cc11c059f95c785900add82e891f69dae9bd977","observation_id":"2e4c03c5-fabf-4c4d-9e39-8e0adf31265c","resolution":{"observed_at":"2026-08-02T07:01:57.727050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:57.786798Z","title":"Fine-grained interpretation of remote sensing image: A review,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:57.786798Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:a80d1a6d1313e313da83f53e6e5a6a9b31d78161882ec6a2f8146f5b4f71dcb8","observation_id":"9295e66c-6f95-4831-997a-77e42f250aa7","resolution":{"observed_at":"2026-08-02T07:01:57.786798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:57.875201Z","title":"A survey on deep learning-driven remote sensing image scene understanding: Scene classification, scene retrieval and scene-guided object detection,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:57.875201Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:0203993c961d426ba627c3e60a25f3b8f42aa30656e212e800588ffa93509d7e","observation_id":"9666adb1-f39a-454d-a5d9-40414da9c6fd","resolution":{"observed_at":"2026-08-02T07:01:57.875201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:57.938841Z","title":"Deep learning for remote sensing image scene classification: A review and meta-analysis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:57.938841Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:721ff90ca5e3d3b19eeeb7b60315eb05b0c29e93a1110685950ad2e592ec6d4f","observation_id":"95cdc4c1-238e-4661-bed2-da24c58a0dc0","resolution":{"observed_at":"2026-08-02T07:01:57.938841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.036163Z","title":"Object detection in optical remote sensing images: A survey and a new benchmark,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.036163Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:1a6b8f4656441f8f478b4b7848995c4dcda1a51112d5d05b697c6a14c4d319f0","observation_id":"8f314a3d-7cb9-487d-b762-c2392d64896c","resolution":{"observed_at":"2026-08-02T07:01:58.036163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.153824Z","title":"Oriented object detection in optical remote sensing images using deep learning: A survey,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.153824Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:6c370166b93577aee62fd0ef4d18a32c4126a2528c8b82dbfbd0e225f42d59bd","observation_id":"7828c2c5-8008-4d89-b087-ea3b453144c4","resolution":{"observed_at":"2026-08-02T07:01:58.153824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.248404Z","title":"A review of remote sensing image segmentation by deep learning methods,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.248404Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:00e82bafd10f1dd5803d21c95fa9e0047d61fcc7891e651b4e09b0a90c1bebfd","observation_id":"ba288dbe-4b35-4025-82e6-a2002f46aea7","resolution":{"observed_at":"2026-08-02T07:01:58.248404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.357159Z","title":"Deep learning-based semantic segmentation of remote sensing images: A review,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.357159Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:3b9dbb443ea4fd7799c0e0235f68a00ecaac8a4f8ebd4a105ca49d27d174e7fe","observation_id":"7ecb6e79-8d2c-48a1-99a9-3fa8cc538116","resolution":{"observed_at":"2026-08-02T07:01:58.357159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.423538Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.423538Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:5ad3166e07dbc09a2ae9416b64908d9a2e70c679d73210906bc47d4111df8de5","observation_id":"6b5c4d1b-6885-4be1-a30c-f3358837e361","resolution":{"observed_at":"2026-08-02T07:01:58.423538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.525812Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.525812Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:1e6bdd6621d52a992c11022569a3153ceaad8bfe7cb88974fd229047b27d0488","observation_id":"4992ddf3-6fcc-4cce-9436-9a877318dc0d","resolution":{"observed_at":"2026-08-02T07:01:58.525812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.584480Z","title":"RS-CLIP: Zero-shot remote sensing scene classification via contrastive vision-language supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.584480Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:d82c722f03d9ddfa7850bb33d2582ae76256d23bf8d6077bd711f6459163b001","observation_id":"1299bebb-d08f-48ec-95cc-64b0588805f8","resolution":{"observed_at":"2026-08-02T07:01:58.584480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.721920Z","title":"RemoteCLIP: A vision-language foundation model for remote sensing,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.721920Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:d90e974a2aafa05e7c8d8b91a8ba26c16d811da23f7903b885fed6de155da8e1","observation_id":"55e07793-2496-4b27-bc43-77bc00b79757","resolution":{"observed_at":"2026-08-02T07:01:58.721920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:58.896410Z","title":"EarthGPT: A universal multi-modal large language model for multi-sensor image comprehension in remote sensing domain,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:58.896410Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:98dc33b77504a6a00537d840a1f3356a2f640583a02d5f10b1da3b6164da64bd","observation_id":"af9bd028-dda0-453d-b90d-78e3d9cd300d","resolution":{"observed_at":"2026-08-02T07:01:58.896410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:59.006262Z","title":"SkyEyeGPT: Unifying remote sensing vision-language tasks via instruction tuning with large language model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.006262Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:c67188f1267470f04af8457a186e2d92de416f16446f975ef2d183b008352e5f","observation_id":"af8c1928-4a16-49be-8c1c-e66d6d69133d","resolution":{"observed_at":"2026-08-02T07:01:59.006262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:59.105956Z","title":"RSVQA: Visual question answering for remote sensing data,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.105956Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:7077b2831d8ad28da1b866b7ad43b79d6f0dbcb93091c27282c5d195f140b56b","observation_id":"046efae5-6bba-4c5e-b20f-c5e84af2137e","resolution":{"observed_at":"2026-08-02T07:01:59.105956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:59.241569Z","title":"EarthVQA: Towards queryable Earth via relational reasoning-based remote sensing visual question answering,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.241569Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:0c4fd6ddc9666ac2193f894c6722116d2b3a3df359de292533ee12007ed2bfcd","observation_id":"9d59454e-e3de-4b60-a8f0-df59e68d4f74","resolution":{"observed_at":"2026-08-02T07:01:59.241569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:59.456765Z","title":"VRSBench: A versatile vision- language benchmark dataset for remote sensing image understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.456765Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:1190b9ec7dc92465a3940af8d85d8fed24ff32a39905a5258b5f52c3a9520d90","observation_id":"3a43bd33-a223-4668-9c71-c6ee091f3354","resolution":{"observed_at":"2026-08-02T07:01:59.456765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:59.561779Z","title":"RSVLM-QA: A benchmark dataset for remote sensing vision language model-based question answering,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.561779Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:ad2f7b68dec9cd8ed4f51f86163095520ffb5dd70365b4eb19de8ea77b7c2cb7","observation_id":"f4e28dca-44fb-4424-9a3e-7f24bff6ad20","resolution":{"observed_at":"2026-08-02T07:01:59.561779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.07045","last_updated":"2026-05-14T08:15:47Z","snapshot_observed_at":"2026-08-01T14:55:13.771477Z","submitted_at":"2026-02-04T08:21:33Z","title":"VLRS-Bench: A Vision-Language Reasoning Benchmark for Remote Sensing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.07045","snapshot_observed_at":"2026-08-02T07:01:59.700299Z","title":"VLRS-Bench: A vision- language reasoning benchmark for remote sensing,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.700299Z"},"links":{"cited_paper":"/paper/2602.07045","citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:9358dd9054129bf0376372604f5bde86f0b194f5bd1670c832a2a597a3b431bf","observation_id":"d19ca235-7b00-4415-9a9c-49a8189b5d6b","resolution":{"observed_at":"2026-08-02T07:01:59.700299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:59.832861Z","title":"OmniEarth: A benchmark for evaluating vision-language models in geospatial tasks,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.832861Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:ac7d66c7f813a0c50bc6103684bdca5417b423a7bf0d684a9f975974d50ce445","observation_id":"69cc916b-3b57-4471-98be-2892899ade28","resolution":{"observed_at":"2026-08-02T07:01:59.832861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:01:59.978058Z","title":"Long-tailed effect study in remote sensing semantic segmentation based on graph kernel principles,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:59.978058Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:1a58fac688cfffe7ae11cfddb8c3d76de51d466de2ade5e4114e544038c61262","observation_id":"dab301c6-7688-4588-8175-1de015c0af13","resolution":{"observed_at":"2026-08-02T07:01:59.978058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:00.165777Z","title":"MAR20: A benchmark for military aircraft recognition in remote sensing images,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:00.165777Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:a9496fc47b6d105860c6a6dd2341fb132213fe21ae85f61ab10fab2a32e02d0a","observation_id":"cf6a0455-4a3d-4ff7-a57b-4e7b02d76b19","resolution":{"observed_at":"2026-08-02T07:02:00.165777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:00.240463Z","title":"A high resolution optical satellite image dataset for ship recognition and some new baselines,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:00.240463Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:96243acae9e3ccc7016493ebb420d00a5e21106a724bec52ea994b8bdbef9abd","observation_id":"36663278-f5e7-4336-a709-6232b8d739eb","resolution":{"observed_at":"2026-08-02T07:02:00.240463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:00.305072Z","title":"Construction and validation of remote sensing image dataset for fine-grained detection of military vehicles,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:00.305072Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:293a0de222d27b8e06399bacf0c98dc3be79d94fd516f75bb8df7e9e4ba87ab6","observation_id":"deb09bf0-980a-4d11-a198-273092eb35d2","resolution":{"observed_at":"2026-08-02T07:02:00.305072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:00.365300Z","title":"A deep learning SAR target classification experiment on MSTAR dataset,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:00.365300Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:72478cf2d3d329c4a5fd407a6bbd7ac6102a3026c6ed52a30353006a7bac5086","observation_id":"91d00b1e-3c16-4855-83c2-31eb4a18498a","resolution":{"observed_at":"2026-08-02T07:02:00.365300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:00.501800Z","title":"Multi-class geospatial object detection and geographic image classification based on collection of part detectors,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:00.501800Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:eefd386e1e10a6f7f3f7494ebaaa036ce1b809d60cb1232c7a0cdc07b81e6b55","observation_id":"f89a95ed-3c54-4031-941d-0b94742ec333","resolution":{"observed_at":"2026-08-02T07:02:00.501800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:00.717837Z","title":"DOTA: A large-scale dataset for object detection in aerial images,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:00.717837Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:bea1b3936fba873b17572e3fe85e9e47bbba0abd3080ad35afef870ef3ec2ca8","observation_id":"6d997772-040e-40f1-bdf9-0c0c9a3a0a1c","resolution":{"observed_at":"2026-08-02T07:02:00.717837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:00.997868Z","title":"SAMChat: Introducing chain-of-thought reasoning and GRPO to a multimodal small language model for small- scale remote sensing,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:00.997868Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:38ca220eb8c831bc91e59b7b786b3236650a7b35b4ccb07b0b034f8fcb5e1540","observation_id":"b29c8413-7baf-434a-9c6d-016222889327","resolution":{"observed_at":"2026-08-02T07:02:00.997868Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:01.127839Z","title":"CHOICE: Benchmarking the remote sensing capabilities of large vision-language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:01.127839Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:23e3794e9e3e3c11d4f25329f16d53d40bc2f91639f3c60a3d0601dbd48454a4","observation_id":"adbc80cd-69d9-4e3a-b790-30e2f2ca77e9","resolution":{"observed_at":"2026-08-02T07:02:01.127839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:01.256996Z","title":"Exploring models and data for remote sensing image caption generation,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:01.256996Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:615d87f8528ffb84999ea57c69175bc2a07f8ca32ec41ee972b0303598c86c2c","observation_id":"9e2621a5-eb70-4ccf-bbd9-20c093c92387","resolution":{"observed_at":"2026-08-02T07:02:01.256996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:01.418671Z","title":"Multimodal large models driven SAR image captioning: A benchmark dataset and baselines,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:01.418671Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:a0d513a106088354cdae7555ccf87bc6e703253301aeee61f70918b53e16c6af","observation_id":"483ca18b-a985-40aa-8661-399adcd48a30","resolution":{"observed_at":"2026-08-02T07:02:01.418671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:01.552940Z","title":"FSAR-Cap: A fine- grained two-stage annotated dataset for SAR image captioning,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:01.552940Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:2c9741948c016c7ebf642397038b7bb5e8196eda45d592916160606b9b9444a7","observation_id":"3ed00b46-d1cf-45a9-ace5-79be214124b0","resolution":{"observed_at":"2026-08-02T07:02:01.552940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:01.754378Z","title":"RS- LLaV A: A large vision-language model for joint captioning and question answering in remote sensing imagery,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:01.754378Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:f1d676b2f47c227ac8041111883185b449a48d4d2645c38a87c0cf7bb860114e","observation_id":"4e6c1022-3db4-4616-9244-491c262ff76a","resolution":{"observed_at":"2026-08-02T07:02:01.754378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:01.896385Z","title":"SAR ship detection dataset (SSDD): Official release and comprehensive data analysis,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:01.896385Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:1ddc0e3f3b9c723d4ad823d2d725e3feddfd2d7214240acdeb0231a91e10d2b4","observation_id":"e78ab11a-7bf5-4026-a5ce-50836bc8fd59","resolution":{"observed_at":"2026-08-02T07:02:01.896385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:02.018184Z","title":"MilChat: A large language model and application for military equipment,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:02.018184Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:6309208111e47e10ef2fc40359d4d57d79a0ecfb8d60b438297f871c4ff5c33e","observation_id":"e01eb5d6-6d6b-4dce-91e7-b1ab4e828580","resolution":{"observed_at":"2026-08-02T07:02:02.018184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:02.176557Z","title":"Battlefield situation awareness using pretrained generative LLM,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:02.176557Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:047066eb11092fee4707c8414f6901e73983a6ea1329035a8ce7c56fa4f9c00e","observation_id":"66277260-4e45-488c-a73a-72ebe9e230dc","resolution":{"observed_at":"2026-08-02T07:02:02.176557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20297","last_updated":"2024-10-27T00:39:24Z","snapshot_observed_at":"2026-07-06T19:40:12.973874Z","submitted_at":"2024-10-27T00:39:24Z","title":"Fine-Tuning and Evaluating Open-Source Large Language Models for the Army Domain","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.20297","snapshot_observed_at":"2026-08-02T07:02:02.360414Z","title":"Fine-tuning and evaluating open-source large language models for the army domain,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:02.360414Z"},"links":{"cited_paper":"/paper/2410.20297","citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:d1c2654ca56c65a1cdea53bc8cf30b23fe50d5ffbd2a10d2d38b52e5ffa86708","observation_id":"1defebdc-146f-4943-ba21-d80582e2a4fc","resolution":{"observed_at":"2026-08-02T07:02:02.360414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:02.454124Z","title":"Deep semantic understanding of high resolution remote sensing image,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:02.454124Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:191e8440e0ed86851e840584552a64942af457062d214e1e41a5a7974d03d9fb","observation_id":"0212d2aa-0b08-4f43-bc22-62cafebbf287","resolution":{"observed_at":"2026-08-02T07:02:02.454124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:02.633655Z","title":"GEOBench-VLM: Benchmarking vision-language models for geospatial tasks,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:02.633655Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:1c9e56d78f4c497280ab261dc33687a6cf1dfd0cdde47968791df62705fd5d76","observation_id":"93f69b87-cf49-4f6c-b6eb-8fb5d8e657e2","resolution":{"observed_at":"2026-08-02T07:02:02.633655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:02.798014Z","title":"RSGPT: A remote sensing vision language model and benchmark,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:02.798014Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:3e6f64e2f715288ddaef8f031650629921706f0412918ff5430d968a3ff83227","observation_id":"308eaf74-6992-4ceb-9f33-97c849270cd0","resolution":{"observed_at":"2026-08-02T07:02:02.798014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:02.926577Z","title":"GeoChat: Grounded large vision-language model for remote sensing,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:02.926577Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:ce9b7183de166f00915067636e4fdf080b182028cbf6ceb133d3c2357326bd62","observation_id":"8dc022b7-a444-47f3-9988-651b848e23af","resolution":{"observed_at":"2026-08-02T07:02:02.926577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:03.048327Z","title":"LHRS-Bot: Em- powering remote sensing with VGI-enhanced large multimodal language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:03.048327Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:fea0b061380fa16c604b3e373a1082c42dca130731a152e4c5a3beaf9f9f9059","observation_id":"e269de29-0b1b-48f8-a140-33d5d64ea681","resolution":{"observed_at":"2026-08-02T07:02:03.048327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:03.259035Z","title":"Introducing GPT-5.4,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:03.259035Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:3d6718bf9d2bcd4c98906fc794bc0bdbe945f2a567582237cc4e791747870589","observation_id":"fd05bbea-6d6f-4416-b874-3e071f60e315","resolution":{"observed_at":"2026-08-02T07:02:03.259035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:03.393022Z","title":"Grounding DINO: Marrying DINO with grounded pre- training for open-set object detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:03.393022Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:5d138016b6acd99edf5284b627570c9949f1229d0d5b42e059c53e8a1f40da3c","observation_id":"af5923d2-a323-4b7e-ab3d-fd2e12722957","resolution":{"observed_at":"2026-08-02T07:02:03.393022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.16719","last_updated":"2026-03-28T16:54:56Z","snapshot_observed_at":"2026-08-07T05:59:39.049027Z","submitted_at":"2025-11-20T18:59:56Z","title":"SAM 3: Segment Anything with Concepts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.16719","snapshot_observed_at":"2026-08-02T07:02:03.521164Z","title":"SAM 3: Segment anything with concepts,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:03.521164Z"},"links":{"cited_paper":"/paper/2511.16719","citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:354a5d09e27e532096230970b8e3b9856ac6925ee0ee930f5604ab580587a01f","observation_id":"0e0901f2-cbe8-4ae0-884a-ba8315c6cc48","resolution":{"observed_at":"2026-08-02T07:02:03.521164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:03.655700Z","title":"Generating plausible distractors for multiple- choice questions via student choice prediction,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:03.655700Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:2c3a6def2717ac62ab8b56175c41694e6665f9d1451d5c7b071e0e18696419f2","observation_id":"1379aeac-22ad-43ce-9e6c-0aba15256823","resolution":{"observed_at":"2026-08-02T07:02:03.655700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:03.805124Z","title":"BLEU: A method for automatic evaluation of machine translation,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:03.805124Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:f45dfc93d06af512b7dc42aadf978ebaac3a21cfde52bc6d8349979dbee348be","observation_id":"f781ae58-78b4-4669-aa15-84d8c24f7be9","resolution":{"observed_at":"2026-08-02T07:02:03.805124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:03.927078Z","title":"ROUGE: A package for automatic evaluation of summaries,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:03.927078Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:b613d8341c958b7da96c30cfc3ecd49a0af2a560bc4a6cfdfbe00cb40897ef1e","observation_id":"f5b0807c-b9fa-4eaa-885e-6ff95888e4d1","resolution":{"observed_at":"2026-08-02T07:02:03.927078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:04.089682Z","title":"BERTScore: Evaluating text generation with BERT,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:04.089682Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:5f9ce8f0a94c61a662ea2aca68344575c273b822559cd09e43ce8569c127e7c0","observation_id":"b36feb9d-f3f4-40a1-b408-5cf32f095b76","resolution":{"observed_at":"2026-08-02T07:02:04.089682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:04.203877Z","title":"The PASCAL visual object classes (VOC) challenge,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:04.203877Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:0c43a4e49d8c250141dd8b978f3d8566e3a01de336ba0beb5a7db51c73a025fa","observation_id":"179dcb5f-3f1d-416d-95dc-f9f28006595e","resolution":{"observed_at":"2026-08-02T07:02:04.203877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:04.320889Z","title":"Natural language object retrieval,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:04.320889Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:f4c25fd227a652d58403830c98c1044d8f24a7aa49f4523b72aabec13002a5d2","observation_id":"58a8ba7a-134e-4dd2-ae43-7665eac60554","resolution":{"observed_at":"2026-08-02T07:02:04.320889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:04.495809Z","title":"Fully convolutional networks for semantic segmentation,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:04.495809Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:85968ba9275c4bf7b108cc5adcf4772e8cc420cf1901936997159eec33ce9484","observation_id":"70d9a255-76f9-474a-baab-bf09952d81e6","resolution":{"observed_at":"2026-08-02T07:02:04.495809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:04.667092Z","title":"Measures of the amount of ecologic association between species,","venue":null,"work_id":null,"year":1945},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:04.667092Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:bd9ee04a531f524c5e3a41d78021d0866e2ad3a3ec8a721699b9193a5db506f0","observation_id":"c6c035ac-b0c7-41cd-9510-d389380fe3a2","resolution":{"observed_at":"2026-08-02T07:02:04.667092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:04.750111Z","title":"Gemini 3 Flash,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:04.750111Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:0e0dab0a492e7df8162d30525c7031e066974baa0e385fd0ec05571ca82a0d81","observation_id":"a86fa36a-8b3c-4f06-9cca-039816ebc2c5","resolution":{"observed_at":"2026-08-02T07:02:04.750111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-02T07:02:04.867362Z","title":"Qwen3-VL technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:04.867362Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:60cf60c90c2d73cf30b50db2052aeb2b78ae996d49b55ce71e5a039e84b18982","observation_id":"aac11fbe-ea5d-4e5e-bbed-4d88122170c5","resolution":{"observed_at":"2026-08-02T07:02:04.867362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:05.012672Z","title":"GPT-4o system card,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:05.012672Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:032a35f594c7cf6fcdece4797aad73f3008a1b037e1e55a64e8583b32f309bbf","observation_id":"8943c0a6-5674-4619-967f-af2f991199ba","resolution":{"observed_at":"2026-08-02T07:02:05.012672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-02T07:02:05.156588Z","title":"Qwen2.5-VL technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:05.156588Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:f045ea588d4f9fa98fbf7a0633edbd7508fd2739fe6de21293316a649adbfb90","observation_id":"dde9a257-683e-4c8b-890f-832cf5367a48","resolution":{"observed_at":"2026-08-02T07:02:05.156588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T07:02:05.336238Z","title":"Claude Sonnet 4.6 system card,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-02T07:02:05.336238Z"},"links":{"citing_paper":"/paper/2607.24810"},"observation_digest":"sha256:b76064f8eb9dd4e0836b8ac9d1fe60e1c156037e396a63bf54d023a06e32e8c8","observation_id":"0a1668fa-a698-46a8-ab74-5f8aeb12a0a1","resolution":{"observed_at":"2026-08-02T07:02:05.336238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.24810","last_updated":"2026-07-13T08:02:57Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-05T20:34:35.919060Z","submitted_at":"2026-07-13T08:02:57Z","title":"RRS-10K: A Multitask Vision-Language Model Benchmark for Rare Remote Sensing Image Interpretation"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":60,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 0 inbound Pith citation observations for arXiv:2607.24810."}