{"as_of":"2026-08-07T18:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7a55e13ce3890f1e2b50c1c4df948c6fb97077d92d36b5d7a36aab52fba42e61","coverage":[{"denominator":53,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T21:47:04.669204Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2602.19190/citation-record","integrity":"/paper/2602.19190/integrity","json":"/paper/2602.19190/citation-record.json","paper":"/paper/2602.19190"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-02T21:47:00.159441Z","title":"Qwen3-vl technical report.arXiv preprint arXiv:2511.21631, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.159441Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:13d3c091330b77a0b4e8c9008cf3024b078b640584c63d7dc568cd42bf1e421a","observation_id":"4a1d9cd5-0a6d-4593-9242-8ac4b7c4089f","resolution":{"observed_at":"2026-08-02T21:47:00.159441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.243122Z","title":"Qwen2.5-vl technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.243122Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:ee88d0b9b1e6cba7ec1dbd90cbf3cd70d9967104eae90385e2af0ca769dc34d3","observation_id":"c1827d55-dcf3-4c0b-943e-13b6e65721fa","resolution":{"observed_at":"2026-08-02T21:47:00.243122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.336448Z","title":"Multi-spectral remote sensing image retrieval using geospatial foundation models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.336448Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:9cc73159ed8cad29b8b6898c46a117ccee95d1d545d7f7eec40bb6a6f33c5651","observation_id":"3b652e5c-b10f-48b8-936a-634a3b9bbb24","resolution":{"observed_at":"2026-08-02T21:47:00.336448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.410631Z","title":"Brown, Michal R","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.410631Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:017e16500346bd0ed39117fa907b7e201bc4972ecc1c031f0228452be780c741","observation_id":"b273c935-9a05-44a3-8c12-5d0dbca40fbc","resolution":{"observed_at":"2026-08-02T21:47:00.410631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.522548Z","title":"Changeclip: Remote sensing change detection with multi- modal vision-language representation learning.ISPRS Jour- nal of Photogrammetry and Remote Sensing, 208:53–69,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.522548Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:5baf75e3539329b4c01ed86c90272c7a4f135166273111aab3904919fe1dd869","observation_id":"87f84470-04b8-4af2-bd58-cd47c4e25988","resolution":{"observed_at":"2026-08-02T21:47:00.522548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.660623Z","title":"Combining sam with limited data for change detection in remote sensing.IEEE Transactions on Geoscience and Remote Sensing, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.660623Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:b36ae0344459278fb80d581289c7decd6c0b8912e296fbfef782b01dffeef666","observation_id":"68c87035-719c-4166-99ad-cb66105ee3b2","resolution":{"observed_at":"2026-08-02T21:47:00.660623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.765883Z","title":"Deep- learning for radar: A survey.IEEE Access, 9:141800– 141818, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.765883Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:18e0ba4ee00e7e5d6f9ea89ba060c6a292abdce0a1e9f1a2e7a9c567898f0c84","observation_id":"8e6191f6-3429-40ae-b4ac-d04e8106f2f8","resolution":{"observed_at":"2026-08-02T21:47:00.765883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.835884Z","title":"Unpaired image-text match- ing via multimodal aligned conceptual knowledge.IEEE Transactions on Pattern Analysis and Machine Intelligence, 47(7):5160–5176, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.835884Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:54b16d800bc81165d8e531d22343773521aa3d61d3612a53e5b164aba2b82a2d","observation_id":"ff677ab4-c154-48ea-98f6-b7fb1b3fe55f","resolution":{"observed_at":"2026-08-02T21:47:00.835884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:00.913150Z","title":"A fast progressive ship detection method for very large full-scene sar images.IEEE Transactions on Geo- science and Remote Sensing, 62:1–15, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:00.913150Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:cdd80cae0b2a0a7b6328e3771e8dab754e2a896cfc433b872281c06e87a67f78","observation_id":"1333c661-a96d-48ed-8c37-e926d8c7622a","resolution":{"observed_at":"2026-08-02T21:47:00.913150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.005862Z","title":"Sarclip: a multimodal foundation framework for sar imagery via con- trastive language-image pre-training.ISPRS Journal of Pho- togrammetry and Remote Sensing, 231:17–34, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.005862Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:4b13a9dfe14ead05ae4d53413c3f1a050f1562a37706e2de4e04c57faf23fac2","observation_id":"f658c8b6-1846-4ea6-91b4-655424f225e8","resolution":{"observed_at":"2026-08-02T21:47:01.005862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.094296Z","title":"Em-yolo: A fine-grained recognition model for aircraft targets in sar im- ages","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.094296Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:47702264386f1e018b97b04a79a5d1533e4fc6fb04bbaf5e409de782a55b1c7d","observation_id":"1097abd9-5ccd-4918-8392-7e27724164c9","resolution":{"observed_at":"2026-08-02T21:47:01.094296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.191491Z","title":"Sar ship detection based on an improved faster r-cnn using deformable convolution","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.191491Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:76ff712d183e4b7cd879d0daf31b462b5e9e9bde2494a12ee41aed33d6cee48f","observation_id":"6fbf4849-38ec-4ddc-91a9-68f123c00731","resolution":{"observed_at":"2026-08-02T21:47:01.191491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.320843Z","title":"Geochat: Grounded large vision-language model for remote sensing","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.320843Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:077dff7c9c3f37556af21db5c5410533d21220b4acff96677ad52a9ff01e2365","observation_id":"dd608eeb-097d-4140-91d0-b0131a9f4319","resolution":{"observed_at":"2026-08-02T21:47:01.320843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.407673Z","title":"Llava-onevision: Easy visual task transfer, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.407673Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:a08582b07f29a57689081b1855d232e77e5376fd1f7ba2dc4d70d3d46210083d","observation_id":"9751cc24-2318-45f1-81d5-29457cb8fdae","resolution":{"observed_at":"2026-08-02T21:47:01.407673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.472873Z","title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.472873Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:c8b94eb142aeb42bd509b29bc7870a1e7eb77852bee2293dd655e257ca9d3f72","observation_id":"e5d53f27-2f39-481a-86ea-8ac61c154189","resolution":{"observed_at":"2026-08-02T21:47:01.472873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.532051Z","title":"A new learn- ing paradigm for foundation model-based remote-sensing change detection.IEEE Transactions on Geoscience and Re- mote Sensing, 62:1–12, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.532051Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:8af21b29aa56f9ad9958e95e678277a7f982b355bbf755207dfe908ec3cbfa64","observation_id":"7311518f-8bce-4cb2-bcc6-c30ccc2b0cb1","resolution":{"observed_at":"2026-08-02T21:47:01.532051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.587363Z","title":"Co-training vision-language models for remote sensing multi-task learning.Remote Sensing, 18(2): 222, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.587363Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:05480e8553cf3fffe5e1bcc93ee52f9fb21adb519cff1c44f576f63241003990","observation_id":"e879ceb5-1487-4b10-8b70-c2a7e1ecd95a","resolution":{"observed_at":"2026-08-02T21:47:01.587363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.671693Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.671693Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:d92ad43fb6c6bc85f4e3b02ca625a54eaa0e4462f4a117571a293b2963f5bbb9","observation_id":"847853ec-32d5-4941-8a1a-5d055eace7ee","resolution":{"observed_at":"2026-08-02T21:47:01.671693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.763737Z","title":"Star: A first-ever dataset and a large-scale benchmark for scene graph generation in large-size satellite imagery.IEEE Trans","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.763737Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:5d0a165fb22d13be45b627b2c15721fbb291c7a22ee8f5cf066560660b8968de","observation_id":"0e0e6b94-402e-40df-ba6a-7a596b1b1ff3","resolution":{"observed_at":"2026-08-02T21:47:01.763737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.872284Z","title":"Re- moteclip: A vision language foundation model for remote sensing.IEEE Transactions on Geoscience and Remote Sensing, 62:1–16, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.872284Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:e13a478018cc0186376c76a5c3a220805b305750506febb8fa33558fef1aaf8b","observation_id":"c9fdb410-9f36-48a5-9c1a-b7a390f4f9c4","resolution":{"observed_at":"2026-08-02T21:47:01.872284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:01.967690Z","title":"Improved baselines with visual instruction tuning, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:01.967690Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:d539813dbb8cec32e15a1ea06f7770d076d2ed78fd8945a455608670492d0660","observation_id":"03b5fdf9-576d-4099-9e96-56769a1dc7b1","resolution":{"observed_at":"2026-08-02T21:47:01.967690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.043624Z","title":"Llava-next: Im- proved reasoning, ocr, and world knowledge, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.043624Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:073c19542bfab7cea8f369faa17d70e0d08d97724687ef77a128d4e99cfb8e63","observation_id":"5032543d-858d-44c0-973a-38b5391247a7","resolution":{"observed_at":"2026-08-02T21:47:02.043624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.107226Z","title":"Texture classification from random features.IEEE transactions on pattern analysis and machine intelligence, 34(3):574–586, 2012","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.107226Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:4ade35e3df229e2694cf74325bc19d19a6f4a7bc22e7b88fc7d922d8d828f259","observation_id":"ddc3ecea-cfcf-48ed-b64b-d11515d66ad4","resolution":{"observed_at":"2026-08-02T21:47:02.107226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.222057Z","title":"A causal adjustment module for debiasing scene graph generation.IEEE Trans- actions on Pattern Analysis and Machine Intelligence, 47(5): 4024–4043, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.222057Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:b5f56c157385c86e7ce4dbaba73f34795944244d787554f7478fc5822b7834c0","observation_id":"e0cd5172-886c-4a57-81e2-e6c552f1d6c2","resolution":{"observed_at":"2026-08-02T21:47:02.222057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.350777Z","title":"Atrnet-star: A large dataset and benchmark to- wards remote sensing object recognition in the wild.IEEE Transactions on Pattern Analysis and Machine Intelligence,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.350777Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:352edbc5b8286d0749bafeedd661259154f4b72eb40b9f221f26b8f105bcb380","observation_id":"c1ff220c-0a5e-48d5-a3a8-aa6cdda9d966","resolution":{"observed_at":"2026-08-02T21:47:02.350777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.475420Z","title":"Vhm: Versatile and honest vision language model for remote sensing image analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.475420Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:6cb89dc93eeeaccf1e2b88265677eaf5def050137b3d8aecfcbb07b6a2104dac","observation_id":"94f33e6d-503a-4e36-8815-f7a98e71e93f","resolution":{"observed_at":"2026-08-02T21:47:02.475420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.578235Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.578235Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:da3093ef2071705ba207f5c3e61d8901fec43f24f31abef39f5914e5bb986bf7","observation_id":"a8237ffe-93fd-4ffe-80e4-be3551e3f696","resolution":{"observed_at":"2026-08-02T21:47:02.578235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.10554","last_updated":"2026-04-02T03:14:28Z","snapshot_observed_at":"2026-07-30T11:23:08.233438Z","submitted_at":"2025-12-11T11:38:50Z","title":"Grounding Everything in Tokens for Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.10554","snapshot_observed_at":"2026-08-02T21:47:02.670792Z","title":"Grounding everything in to- kens for multimodal large language models.arXiv preprint arXiv:2512.10554, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.670792Z"},"links":{"cited_paper":"/paper/2512.10554","citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:bdeb30646dc417ce2f6d5ebebc3e0ea74fd4931babb4fb9401f8a1c7232c39e3","observation_id":"30ee7653-332c-433a-8396-7340f9a605da","resolution":{"observed_at":"2026-08-02T21:47:02.670792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.720036Z","title":"Earthdial: Turning multi-sensory earth observations to interactive dialogues","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.720036Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:59f69c4122b7174884ec1933be13c715639c40baf2eb739cfbca291e5d0abad5","observation_id":"2636d552-4451-4f70-91ad-669b5defe258","resolution":{"observed_at":"2026-08-02T21:47:02.720036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.806763Z","title":"Fully polsar image reconstruction for enhanced land cover mapping.Pattern Recognition, 169:111895, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.806763Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:c2e7dfbf05dbbab435a4ee80f31e311608c0b914cea2d395c9e01174b03d668e","observation_id":"96b05514-31a5-43ba-9864-c4080fbe7bfb","resolution":{"observed_at":"2026-08-02T21:47:02.806763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.865615Z","title":"Vi- sual position prompt for mllm based visual grounding.IEEE Transactions on Multimedia, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.865615Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:39eff386b4ed2b91072be82dd46b872b2e503281a83c5088229c22785eaef952","observation_id":"65ae49c1-3117-4d2d-8a93-1474b238facf","resolution":{"observed_at":"2026-08-02T21:47:02.865615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:02.926297Z","title":"Re- ferring expressions as a lens into spatial language grounding in vision-language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:02.926297Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:7a15c0f3b2a75293c75303f434d33a85cc46b4e8b58169fb0f5ea601069b8b90","observation_id":"82e796d5-c171-4dcc-957d-5ec0b4323b50","resolution":{"observed_at":"2026-08-02T21:47:02.926297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.020406Z","title":"Group equivariant u-net for the semantic segmentation of sar im- ages","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.020406Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:7f11ddd5ee1059217c3013cbf2b2cabc415f50e57373065d5d040b6b2bfcfeac","observation_id":"e5f1c877-975e-4832-8af8-9d7fa80254c8","resolution":{"observed_at":"2026-08-02T21:47:03.020406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.077134Z","title":"Annotation-free, high-fidelity sar oil-spill image synthesis via classification-guided diffusion model.IEEE Transactions on Geoscience and Remote Sens- ing, 63:1–11, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.077134Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:c69720c729e1a518b9dc71a42bcb19c762f2d545614d4c70f43a0fbf6ede7ae3","observation_id":"5f93fb2f-fb8d-416b-b3ab-ebae82e92f7b","resolution":{"observed_at":"2026-08-02T21:47:03.077134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.159165Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.159165Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:4fcc42d8d956f20ecac8f4ea17bc450aa5e68bbc7bccfdf02684190e8f439b41","observation_id":"c106161f-313b-42c0-af97-81130340cb03","resolution":{"observed_at":"2026-08-02T21:47:03.159165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.18265","snapshot_observed_at":"2026-08-02T21:47:03.238957Z","title":"Internvl3.5: Advancing open-source multimodal models in versatility, reasoning, and efficiency","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.238957Z"},"links":{"cited_paper":"/paper/2508.18265","citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:ecdc6ebcbbf538dd2bb335ba571fa4301a1198c1efd36884f257b3fb676e9423","observation_id":"7219751a-a3fc-4116-9c8a-9f13a82b205e","resolution":{"observed_at":"2026-08-02T21:47:03.238957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.364687Z","title":"Skyscript: A large and seman- tically diverse vision-language dataset for remote sensing","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.364687Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:f2d1dc9fb3ecb81ee0c8b3140b92e152fe472ef7e0c7e1c6e506f705767c196a","observation_id":"40e3341b-a7dd-43d5-bcae-e3e217765fad","resolution":{"observed_at":"2026-08-02T21:47:03.364687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.430443Z","title":"Sarlang-1m: A benchmark for vision-language modeling in sar image un- derstanding.IEEE Transactions on Geoscience and Remote Sensing, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.430443Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:1611d2c2eb1fe7ee7167efeefb94e9f3ec9186d86e26c001f5b607891ca01f13","observation_id":"f4c42a91-38f8-494f-850e-962fcfcf5088","resolution":{"observed_at":"2026-08-02T21:47:03.430443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.497183Z","title":"Bootstrapping interactive image–text alignment for remote sensing image captioning.IEEE Transactions on Geoscience and Remote Sensing, 62:1–12, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.497183Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:6bce5695011e8d095fe1d4d91f3f9da511c056deab58c14165fc5d54f40fc4a6","observation_id":"8e2d6445-020a-4461-a9a0-9fe298599595","resolution":{"observed_at":"2026-08-02T21:47:03.497183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.564279Z","title":"R3det: Refined single-stage detector with feature refinement for ro- tating object","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.564279Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:f646e2682ef60dc765861986befb0c629c34fd6d21d8289e87a736fe2f9b54e9","observation_id":"7278e134-5a14-46b5-9afc-5348ab33b4c4","resolution":{"observed_at":"2026-08-02T21:47:03.564279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.659956Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.659956Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:c58904befd0fc1aa618cbf3289838d77d683910256345ccf06d868e32c3344c1","observation_id":"65a717dc-5683-4db1-9176-8ce0a9c40e60","resolution":{"observed_at":"2026-08-02T21:47:03.659956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.766569Z","title":"Fusar-klip: Towards multimodal foundation models for remote sensing, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.766569Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:5dd926e74b46794af1774c7651eb406d9357e8c1ca8b9db574dccf1036ae6db6","observation_id":"3f3da96d-90ac-4bee-86a2-de6914c4fedf","resolution":{"observed_at":"2026-08-02T21:47:03.766569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.820210Z","title":"Object fidelity diffu- sion for remote sensing image generation.arXiv preprint arXiv:2508.10801, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.820210Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:16a0b082da786b63cbf26236377e7e02599ee9ad40e1ec7178899412150b9601","observation_id":"8ae3e397-d85e-4bf1-99c4-058d6ad952f2","resolution":{"observed_at":"2026-08-02T21:47:03.820210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.888503Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.888503Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:62247b2b8a64597784cb1d7c8611c3bc7bf4e9a10596bed271bfaab55083c1fb","observation_id":"ae798bc3-940e-4d09-a463-f9e2b0589c94","resolution":{"observed_at":"2026-08-02T21:47:03.888503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:03.977426Z","title":"Earthgpt-x: A spatial mllm for multi-level multi-source re- mote sensing imagery understanding with visual prompting,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:03.977426Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:4a52f29c34eeaf72128f32397d84331808a8fe90b6bca63076544973ee87b85a","observation_id":"51e2950d-11f2-497b-9fd0-a9c76a72eac4","resolution":{"observed_at":"2026-08-02T21:47:03.977426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.061253Z","title":"A fast training method for sar large scale samples based on cnn for targets recognition","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.061253Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:ebd578d921f3e1f38481871ba3acf5924b6ee81f33d26bda49521a736bdaeeb4","observation_id":"a035c27e-c4e5-4ddd-9e5c-80ce78f2d639","resolution":{"observed_at":"2026-08-02T21:47:04.061253Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.164672Z","title":"Rs5m and georsclip: A large-scale vision-language dataset and a large vision-language model for remote sensing.IEEE Transactions on Geoscience and Remote Sensing, 62:1–23,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.164672Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:7d6aad168d9da07629bb9f05c1561d25b4bea93375678d15b2ae538a7023dc9d","observation_id":"affe2e7f-db1b-4d46-b862-806b358bf36c","resolution":{"observed_at":"2026-08-02T21:47:04.164672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.259014Z","title":"Geo-r1: Improving few-shot geospatial referring expression understanding with reinforcement fine- tuning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.259014Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:9f8f7d032374b2bf919f19009e98f1068dd6d93e8c64383d6085e0b16bc9e1c3","observation_id":"4b761e81-d0b8-4935-9497-958a3d42c5fd","resolution":{"observed_at":"2026-08-02T21:47:04.259014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.363440Z","title":"Towards vision- language geo-foundation model: A survey.arXiv preprint arXiv:2406.09385, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.363440Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:7e32b5099b6eaee864b1e5ec67cabd410a845a8eff81cce846a95588e623ff2b","observation_id":"c7ca8b52-d712-4670-a9e3-8dd4b11d2891","resolution":{"observed_at":"2026-08-02T21:47:04.363440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.441518Z","title":"6, SAR imagery exhibits inherent lim- itations that constrain visual–semantic understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.441518Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:0ab259e29129d91b1af3a38a5b658d08fb651d716d1ccfb9fe611749a6414e28","observation_id":"ca862e42-4fd8-44d8-98e6-2d9ab1192f37","resolution":{"observed_at":"2026-08-02T21:47:04.441518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.531930Z","title":"FUSAR-GPT consistently outperforms all competing methods by a signif- icant margin across counting, grid-based localization, and classification tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.531930Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:a9c841fa6da7ead1d3615fe178159a68b76f86e488d64abf2439033fe5c02758","observation_id":"1110d89e-08a1-4980-a8e7-9da1ed3e5c91","resolution":{"observed_at":"2026-08-02T21:47:04.531930Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.613602Z","title":"Ablation results on the target counting task","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.613602Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:dbf1c895a3ccb621055f411234cadc60b48c0a0e20f79c99c59a1265f268f0a8","observation_id":"b4d696b9-b1ea-42fe-a303-415949e6328e","resolution":{"observed_at":"2026-08-02T21:47:04.613602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:47:04.669204Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T21:47:04.669204Z"},"links":{"citing_paper":"/paper/2602.19190"},"observation_digest":"sha256:cf42b9d81a9d802fae587b75e3af75eea7d2aba2ecc302fd45c332232b2ede69","observation_id":"bbf26c0e-e3ff-4fd3-a6e4-530933cdbcb8","resolution":{"observed_at":"2026-08-02T21:47:04.669204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2602.19190","last_updated":"2026-06-04T06:50:26Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T21:46:59.562400Z","submitted_at":"2026-02-22T13:40:17Z","title":"FUSAR-GPT : A Spatiotemporal Feature-Embedded and Two-Stage Decoupled Visual Language Model for SAR Imagery"},"reference_resolution":{"displayed":53,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":52,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":53},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 53 of 53 outbound references and 0 inbound Pith citation observations for arXiv:2602.19190."}