{"as_of":"2026-08-13T06:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:14746405dd5559b577a6db01127034afd9f9ad232c7288316b339bc6d0dc7981","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T15:05:57.863886Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T00:02:12.727566Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-10T00:29:47.839616Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"cited_work":{"arxiv_id":"2411.14704","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.14704","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"93bf4b2b-de35-4668-a83d-bda0174158d7","year":2024},"citing_paper":{"arxiv_id":"2604.20429","last_updated":"2026-04-22T10:50:38Z","snapshot_observed_at":"2026-08-11T14:47:34.636623Z","submitted_at":"2026-04-22T10:50:38Z","title":"Fast-then-Fine: A Two-Stage Framework with Multi-Granular Representation for Cross-Modal Retrieval in Remote Sensing","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T00:02:12.727566Z"},"links":{"cited_paper":"/paper/2411.14704","citing_paper":"/paper/2604.20429"},"observation_digest":"sha256:c55f6353f728fe84efd996c2518928ce9a1ad1dc289618992d06ba7501a17bce","observation_id":"be331e05-aba8-4882-8bdd-c298a7fd228c","resolution":{"observed_at":"2026-05-10T00:29:47.841047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.14704/citation-record","integrity":"/paper/2411.14704/integrity","json":"/paper/2411.14704/citation-record.json","paper":"/paper/2411.14704"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.607767Z","title":"Remote sensing big data computing: Challenges and opportunities,","venue":null,"work_id":"6e7436d9-351d-4695-8637-e4daf60c0741","year":2015},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.642747Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:caf3085f1d89c7cdb969afeca768da8bf27fe6d39a7871ec7ebcb54fe75e2794","observation_id":"4c3c0f33-d76b-44b2-828f-248f4bb5975a","resolution":{"observed_at":"2026-08-12T15:05:58.612459Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.594508Z","title":"Big data for remote sensing: Challenges and opportunities,","venue":null,"work_id":"fc6c6dbe-dbbc-49f6-bc26-c552d83d8f79","year":2016},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.648140Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:cad551714d196092cc31277565ba4efd5908995684149967ea75899c820a64ba","observation_id":"b520db60-7253-48a9-ab56-357ceba5605a","resolution":{"observed_at":"2026-08-12T15:05:58.599090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.578302Z","title":"Understanding urban landuse from the above and ground perspectives: A deep learning, multi- modal solution,","venue":null,"work_id":"6c9c1fa7-fc18-42e8-8c70-6944f6d0d316","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.653137Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:047e71517bd65f7067767491c2f88e02f45e31d4dd76a9caa9bfd9cf3d0bfa7d","observation_id":"186ad7d1-a193-4143-a9b6-b76fada333c3","resolution":{"observed_at":"2026-08-12T15:05:58.583419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.564573Z","title":"Hyperspectral data analysis for arid vegetation species: Smart & sustainable growth,","venue":null,"work_id":"1075291d-9195-4664-8c0b-2323edae402a","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.660856Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:f3ae3ec044a2fa04edebea3565d6a663bfd5e8fc546c6a4ebfd4550b46388e39","observation_id":"43844814-7191-468d-a558-7acd3ab66674","resolution":{"observed_at":"2026-08-12T15:05:58.569310Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.549429Z","title":"Google earth engine cloud computing platform for remote sensing big data applications: A comprehensive review,","venue":null,"work_id":"d9b39105-ab50-481c-9db5-f8fd36e7231e","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.665468Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:63e0aeddafad3b050782be059eb26e50cfbfc3b3d480871e1e748cb20eae78b1","observation_id":"6fa11a44-fe85-41c9-a49a-5b0c45c4ce5f","resolution":{"observed_at":"2026-08-12T15:05:58.554913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.669876Z","title":"Nwpu- captions dataset and mlca-net for remote sensing image captioning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.669876Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:99161f8d025d28bb04b7dceafb634ac40abcf6ebfdbe4ef44736b3dd1a485da6","observation_id":"960bfcb1-ca03-4497-8df8-27754115cfda","resolution":{"observed_at":"2026-08-12T15:05:57.669876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.516916Z","title":"Textrs: Deep bidirectional triplet network for matching text to remote sensing images,","venue":null,"work_id":"bcb69c45-7878-4c9a-a2f6-e9ca8e320772","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.674957Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:8a012f18700960056aee8bdad7d0c0601e4eb961e6ab64cff5a34e5627ced47c","observation_id":"654ef5dd-7f5b-4121-bd63-54b3ae71712e","resolution":{"observed_at":"2026-08-12T15:05:58.521644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.679755Z","title":"A deep semantic alignment network for the cross-modal image-text retrieval in remote sensing,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.679755Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:bdd4592fffba16b763d572301033544aa5a270d24f823b11f13f59b1da8dd3df","observation_id":"313642ac-f437-493f-823a-de51ef14e235","resolution":{"observed_at":"2026-08-12T15:05:57.679755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.494798Z","title":"Fusion-based correlation learning model for cross-modal remote sensing image retrieval,","venue":null,"work_id":"eb96805f-ede1-43df-b506-c46b2b5aebf3","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.683688Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:5e9a1c83716ebc8f8d6528804712bb770650083ab497f4f6abd9c3358735ab68","observation_id":"0a042db3-5854-4d0e-9c98-8f731f04820f","resolution":{"observed_at":"2026-08-12T15:05:58.499246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.481855Z","title":"Cross spectral image reconstruction using a deep guided neural network,","venue":null,"work_id":"e5518036-7646-4035-a66e-e83528ee2563","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.687271Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:145a2ca5843c41c493a005d1bb53c7b417aa9e6d081753eed99706f7768bf91f","observation_id":"b1447892-a6cf-4ffb-84d5-2d9dfe23f503","resolution":{"observed_at":"2026-08-12T15:05:58.486375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.465277Z","title":"Image super-resolution using t-tetromino pixels,","venue":null,"work_id":"d3627432-4bb6-4e45-bf7b-ed46b4fdddbe","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.691404Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:9d0ededc82e151e97c7dcf44f3a0d3f9bf9fa338d22eb1de8068958cb9685530","observation_id":"f87b27fd-5a3f-4f20-b745-98abb2d47e4f","resolution":{"observed_at":"2026-08-12T15:05:58.472800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.695460Z","title":"Exploring a fine-grained multiscale method for cross-modal remote sensing image retrieval,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.695460Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:76eceaecc20ed32f61b3495c9e376574fa0cd0df2ee9659db49f07162a8ccd21","observation_id":"a2bafe0a-f97a-414e-9580-6e01c0556ead","resolution":{"observed_at":"2026-08-12T15:05:57.695460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.699844Z","title":"Remote sensing cross-modal text-image retrieval based on global and local information,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.699844Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:3be4263ea9fa45296ae44e1a5d80311500c7532ef75ff20a07df2bfa2a7f246f","observation_id":"deb1efde-12a0-4211-994d-3d1cc98e2126","resolution":{"observed_at":"2026-08-12T15:05:57.699844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.436486Z","title":"A lightweight multi-scale crossmodal text-image retrieval method in remote sensing,","venue":null,"work_id":"120d1daf-b135-4bd4-ac3b-36f749c9205b","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.704005Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:8750492776abf3f6e4901a55ee4cb276adaa2e67329c0e78bc98063206e63c02","observation_id":"14cdfa4f-a120-4b0e-ae72-c3c707dcabf0","resolution":{"observed_at":"2026-08-12T15:05:58.440911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.422609Z","title":"Exploring uni-modal feature learning on entities and relations for remote sensing cross-modal text-image re- trieval,","venue":null,"work_id":"c0cab0b8-e3f6-4852-b6e9-d360edf1214f","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.707863Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:0a3db26853dbf705a739ebfed84d921828c37d77b89b8f1246b627040cb9f68a","observation_id":"c8db1e29-8418-4854-ae4c-169435be8548","resolution":{"observed_at":"2026-08-12T15:05:58.427872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.409098Z","title":"Hypersphere-based remote sensing cross-modal text-image retrieval via curriculum learning,","venue":null,"work_id":"bedb3578-a6ae-4b4f-b16b-4729279c61ed","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.711725Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:eec825681f6fbab0ef46a395a333f5355dcea552ffc76633613a145ed37567ca","observation_id":"70b6b766-acec-470d-b78a-489c3a11e6af","resolution":{"observed_at":"2026-08-12T15:05:58.413683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T02:40:23.887636Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T15:05:57.715976Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.715976Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:effda024b4ad0d2cfdf169280417bc1938cb273e834649b7a9476ff27367bf4f","observation_id":"d19e02e5-e545-43e2-b5ad-b9799914777c","resolution":{"observed_at":"2026-08-12T15:05:57.715976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.394847Z","title":"Long short-term memory recurrent neural network architectures for large scale acoustic modeling,","venue":null,"work_id":"f713e819-382c-4aff-aa25-cc38e421c731","year":2014},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.720276Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:9fce721ce4a8fcfb42e3a69617afabdf1eb17f52357c1de51e1b774efbc4b93b","observation_id":"4f20a153-5c0f-4009-a05c-947e74b345ca","resolution":{"observed_at":"2026-08-12T15:05:58.399313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.3555","last_updated":"2014-12-11T06:46:53Z","snapshot_observed_at":"2026-08-10T04:10:43.257455Z","submitted_at":"2014-12-11T06:46:53Z","title":"Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.3555","snapshot_observed_at":"2026-08-12T15:05:57.724413Z","title":"Empirical evalua- tion of gated recurrent neural networks on sequence modeling,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.724413Z"},"links":{"cited_paper":"/paper/1412.3555","citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:fc5f1f09b99e5908661eeda54c103e0c78735a93c9beb4b0c736da406d55c7a4","observation_id":"beb708e3-24d7-4cce-ae2e-c8810b2f1c7d","resolution":{"observed_at":"2026-08-12T15:05:57.724413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.728685Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.728685Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:227e4a93c5b651158af3481c29cbb457d1f7cc563652dd6a7c8734e0cbaa7497","observation_id":"6dce598e-f33e-42cb-8ffc-3cf4bb0e6501","resolution":{"observed_at":"2026-08-12T15:05:57.728685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.373568Z","title":"Multiscale salient alignment learning for remote sensing image-text retrieval,","venue":null,"work_id":"e97334ea-4945-4e26-a4f7-9e9767048db9","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.732957Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:ee172aae5e4f1e8e141509288e3d15f524138f8d943762b7f998b91912265959","observation_id":"dce4e68e-4451-4a12-980a-ede130182e20","resolution":{"observed_at":"2026-08-12T15:05:58.378152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.359007Z","title":"Interacting- enhancing feature transformer for cross-modal remote sensing image and text retrieval,","venue":null,"work_id":"fb43b4e8-47ca-467a-84a9-5ca9455dd452","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.736707Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:34445c7fe8f796223cd49a879c05721b18331047960877ecf0ae8b14f2a483dc","observation_id":"71c63976-4336-441b-8ff7-2f40c3b50181","resolution":{"observed_at":"2026-08-12T15:05:58.364329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.342688Z","title":"Align before fuse: Vision and language representation learning with momentum distillation,","venue":null,"work_id":"daf655d9-d416-4c67-b440-f53f89ca8484","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.740423Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:b8248e84430a7d3cc758b2de86973d8387ebed7c268341b09e5766ede8fda7af","observation_id":"9465be24-0dca-4812-a1c8-70fdddfadfc5","resolution":{"observed_at":"2026-08-12T15:05:58.349930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.327586Z","title":"Deep saliency smoothing hashing for drone image retrieval,","venue":null,"work_id":"6a379627-6175-4b1a-a4f1-1e9fc6020333","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.744193Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:02fe0f3f1b4c6c1bda960792b9924dfea12d1803895ed3054fc9ab3789b16fee","observation_id":"e6a07d92-cebb-495a-8826-6c0c53a5ed8c","resolution":{"observed_at":"2026-08-12T15:05:58.332354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.310177Z","title":"Multitask learning for sar ship detection with gaussian-mask joint segmentation,","venue":null,"work_id":"a3153b2c-bbbc-4cc9-8e8d-f94acf9ccaa7","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.748050Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:28dad47b8bb03b74f61afc71d58b09043afb572ea60a17b55a6380d1f6c9d752","observation_id":"76ed2888-dad4-4d5a-9698-91351679a81c","resolution":{"observed_at":"2026-08-12T15:05:58.314377Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.297962Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows,","venue":null,"work_id":"fee7a107-37db-4861-9bab-38cfe6f337d8","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.751703Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:1f497eb1a053d65a524ae89c4a28e3460803c2d83f0e3514af70a5e2c72c7a75","observation_id":"930a1883-858a-468c-ae0d-e8757f9c2a18","resolution":{"observed_at":"2026-08-12T15:05:58.302154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.285398Z","title":"Global context vision transformers,","venue":null,"work_id":"98cb2788-07d2-4483-a87b-c6ce409971d1","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.755719Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:ac304e176a9192767be6fb463cc81c734be3d5813ed0e3178e2873551cc1bac8","observation_id":"2380a155-4554-40be-b0f7-8cda192742e4","resolution":{"observed_at":"2026-08-12T15:05:58.289581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.272443Z","title":"Matching images and text with multi-modal tensor fusion and re- ranking,","venue":null,"work_id":"9c5cbf1b-22d7-4d4a-a832-a30426490d22","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.759505Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:a88eb6e04a7ace3fddf29e8b9bdb554d6b1f52fc6ab83705ba1f21ca4a6f8942","observation_id":"37aa4e64-5910-4588-9fbb-cf1ea37451dc","resolution":{"observed_at":"2026-08-12T15:05:58.276858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.259617Z","title":"Vse++: Improving visual-semantic embeddings with hard negatives,","venue":null,"work_id":"d2389990-e818-4c30-a041-f66cf1104a26","year":2017},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.763722Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:3e03dd126a84b1d0fad5eb88a70c21a02466525aca329833777245914686e7c5","observation_id":"6d79f654-7652-4f77-a091-b4447c4ddbfb","resolution":{"observed_at":"2026-08-12T15:05:58.264258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.767188Z","title":"Exploring models and data for remote sensing image caption generation,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.767188Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:a656f95feef5907a5b1e90d9daacae30d725a0cbb53e0823a1fcab03b34be020","observation_id":"84ac96ff-b178-43ae-82b3-893433494fb9","resolution":{"observed_at":"2026-08-12T15:05:57.767188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.236889Z","title":"Deep semantic understanding of high resolution remote sensing image,","venue":null,"work_id":"ffd4bd9b-98a9-4b47-876b-d12733fa7d46","year":2016},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.770808Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:d06467455123258e2650594b3836672f30f82bd1b6b2301a604646c60f31fc0e","observation_id":"fafb1176-24c3-427d-b2a1-e7c343df78e3","resolution":{"observed_at":"2026-08-12T15:05:58.242536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.220557Z","title":"End-to-end convolutional semantic embeddings,","venue":null,"work_id":"f7927f39-8dca-402b-8ced-9a11fdcddd2b","year":2018},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.774534Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:ea9465e5fea8ff11c5b4884e417af373ad36870804ea3c670530f2dc7afb19f6","observation_id":"44d0bb0b-68d2-4af5-bf61-64699935b562","resolution":{"observed_at":"2026-08-12T15:05:58.225883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.205158Z","title":"Cross-modal semantic correlation learning by bi-cnn network,","venue":null,"work_id":"a56cfe1f-47c2-4e44-99a5-ea42d1dccf4a","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.778748Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:9bd5216091712294b9e392d66b22d9cfd7b93882268c3210ab09fe91210abec9","observation_id":"519cb981-5d09-44c6-896e-a8e9aecc3cd6","resolution":{"observed_at":"2026-08-12T15:05:58.210444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.190767Z","title":"Dual-path convolutional image-text embeddings with instance loss,","venue":null,"work_id":"8a611450-3c4f-4444-8c27-9e47c0619f31","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.782641Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:f68f8f5fe131f9647b4271eae5fd4ae858744dbbc752cc3f6e54210d73f763c8","observation_id":"f110aff3-2c77-467e-856d-caf394a2b75a","resolution":{"observed_at":"2026-08-12T15:05:58.195285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.786515Z","title":"Deep supervised cross-modal retrieval,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.786515Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:15e4daa2a8519125179853059eb5fb6f6037a7c93a3eeae5db1bd895bfbc387f","observation_id":"e1216bf4-dd3e-4a46-a799-90b70d2aa7f3","resolution":{"observed_at":"2026-08-12T15:05:57.786515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.165509Z","title":"Learning semantic concepts and order for image and sentence matching,","venue":null,"work_id":"9ac638d8-0d58-4ed1-a1e8-66991a749d08","year":2018},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.790384Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:f2667edf4a66c79f7e88daaceab7de2c0d36113bf2647c11b8e4a3c072859b36","observation_id":"03dc5032-001b-4cd2-9d2d-27ef02950545","resolution":{"observed_at":"2026-08-12T15:05:58.170650Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.151060Z","title":"Stacked cross attention for image-text matching,","venue":null,"work_id":"d021d660-aaef-4269-af43-263a2c1d4796","year":2018},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.794558Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:9834c3d70b5e2d1bfe95f8c3c0a57f1ee9b7b341c6bedb5438d09960f56ad26f","observation_id":"ee5e49a4-9c8a-4ed9-be2d-5070c4577c11","resolution":{"observed_at":"2026-08-12T15:05:58.156138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.138030Z","title":"Cross- modal attention with semantic consistence for image–text matching,","venue":null,"work_id":"2fadf9fa-31a4-48a5-8af4-b6db11f750a6","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.798448Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:205882cc9960cee3bbbcc3e37501446836072fc2d83c88da9c31df77a2157ca7","observation_id":"8fd228ad-e85b-43c2-a565-2c2e4aa3d91a","resolution":{"observed_at":"2026-08-12T15:05:58.142074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.125323Z","title":"Visual semantic reasoning for image-text matching,","venue":null,"work_id":"1ea83aa3-014f-478d-be92-dbf13943aee4","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.802192Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:aee2c99bdb9c786adae12e860ab122f98f90aae33ff95d284ef55e350030c07f","observation_id":"38ae2995-86c3-4b14-93b2-da73ff3a6f7f","resolution":{"observed_at":"2026-08-12T15:05:58.129587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.110281Z","title":"Image-text embedding learning via visual and textual semantic reasoning,","venue":null,"work_id":"1416bc64-2da6-4d83-b3b4-979685549c14","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.805867Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:8a64a8150956850e91b88ba6cb8acacb5616f359ca2018a5ee2546eeef44ce51","observation_id":"91c20eb1-1199-455f-a128-3099ece39e98","resolution":{"observed_at":"2026-08-12T15:05:58.115153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.809534Z","title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.809534Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:02e5a285203e8f6ce48f882d77a1b1b1d4ca199ded39a49da757b1a56acc21fa","observation_id":"1414cfaf-f4aa-449e-a1a5-f582b8bb01c7","resolution":{"observed_at":"2026-08-12T15:05:57.809534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.088823Z","title":"Lxmert: Learning cross-modality encoder representations from transformers,","venue":null,"work_id":"313b043c-e4bd-4613-b889-72722d8caae0","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.813277Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:8475a359cada42afc08d7b7689becf2559b19b67f5e9952a2a9c82c16a717c5f","observation_id":"ecf46abb-c98e-4236-af91-6bb7d5159778","resolution":{"observed_at":"2026-08-12T15:05:58.093662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.075571Z","title":"Fashionbert: Text and image matching with adaptive loss for cross- modal retrieval,","venue":null,"work_id":"fc46efbb-b0a0-477b-86bd-f7ab1b4b4138","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.816997Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:3f7253063f533f89d41c9c78f387b203d041a20e16a7674745ea18ab4917e86d","observation_id":"c1a80255-d52c-4e74-87fb-1c654dfcc07f","resolution":{"observed_at":"2026-08-12T15:05:58.080079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.062521Z","title":"Learning the best pooling strategy for visual semantic embedding,","venue":null,"work_id":"320c26eb-158d-4502-b241-02bbed3e9058","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.820850Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:e31c73f5984147ae756db14561ea23dbf843d49dd07cc2f5f05aee11b87298a5","observation_id":"84f014d7-3af0-4263-b883-94ef99a55daa","resolution":{"observed_at":"2026-08-12T15:05:58.066838Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.00849","last_updated":"2020-06-22T09:09:22Z","snapshot_observed_at":"2026-08-04T21:04:33.037459Z","submitted_at":"2020-04-02T07:39:28Z","title":"Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.00849","snapshot_observed_at":"2026-08-12T15:05:57.824935Z","title":"Pixel-bert: Aligning image pixels with text by deep multi-modal transformers,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.824935Z"},"links":{"cited_paper":"/paper/2004.00849","citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:1f17638afaed7953cfab4912910f1add303da1ece98d4a25b34216b4f1f22d5e","observation_id":"0e90948c-19b9-4e50-8e10-f5988d1ed44f","resolution":{"observed_at":"2026-08-12T15:05:57.824935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.829050Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.829050Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:5f4f3064e4cb8b7a0d3a8392b42d458d8b1ea7b0abeb7170e8b6a7e91632db2a","observation_id":"bf64ed5d-685c-43e1-9734-415474ed2ece","resolution":{"observed_at":"2026-08-12T15:05:57.829050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.037721Z","title":"Vista: Vision and scene text aggregation for cross-modal retrieval,","venue":null,"work_id":"9ac05722-6a40-4822-a2b1-c2a473f353f1","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.832696Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:ad252fcefbd1f34254caa193eec690f4ad9748c4ee892a0d194e5489b43d93da","observation_id":"e2c38548-073c-429f-a7d2-c76397f0bc7a","resolution":{"observed_at":"2026-08-12T15:05:58.043211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.022875Z","title":"Vilt: Vision-and-language transformer without convolution or region supervision,","venue":null,"work_id":"189c5315-e747-4766-8d58-946278e7d254","year":2021},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.836567Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:318cd5159cb718313e04e930e23a89e7ddac189eeeda80c3d8891610cf9b85a6","observation_id":"d6f8ddff-9a5c-43b6-82b6-5bc90cca8ffb","resolution":{"observed_at":"2026-08-12T15:05:58.027761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:58.008687Z","title":"Vlmo: Unified vision-language pre-training with mixture-of-modality-experts,","venue":null,"work_id":"ae3f342f-a34a-4fb1-acae-407294eb1e17","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.840435Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:addf3cbae85f4a00b043c5810b5001fabe4f5ed03d3bfbd0fb7ca88247b5de93","observation_id":"597fa9db-e019-4720-8555-7da321fee29a","resolution":{"observed_at":"2026-08-12T15:05:58.013236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.991591Z","title":"Knowledge-aided momentum contrastive learning for remote-sensing image text retrieval,","venue":null,"work_id":"180fefb7-2b82-466e-9907-ef97de88fbb2","year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.844283Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:a6b5d309dd69b5e51912629fa586f1106784388750a797d4676b816c0b0bd716","observation_id":"45903d78-eded-4900-a047-c4855e2cfac6","resolution":{"observed_at":"2026-08-12T15:05:57.999616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.978437Z","title":"Multi- scale interactive transformer for remote sensing cross-modal image-text retrieval,","venue":null,"work_id":"f97777ea-e345-456b-bb14-ac20d28d7679","year":2022},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.848175Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:60b5944d16f89b4fa2501023d760afb05dc3fd57cd910d515607fc0765286a36","observation_id":"9b3afb81-747a-406f-858f-61a9b7dd3591","resolution":{"observed_at":"2026-08-12T15:05:57.983362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.852135Z","title":"Parameter-efficient transfer learning for remote sensing image-text retrieval,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.852135Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:48e24a4457678375cf6e7467c4fb9fa6a8dc0060c602816677beb5c72cd83104","observation_id":"87a56034-0031-4957-9722-3c6a38a28750","resolution":{"observed_at":"2026-08-12T15:05:57.852135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.856091Z","title":"Integrating multisubspace joint learning with multilevel guidance for cross-modal retrieval of remote sensing images,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.856091Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:0399bd7c79693d99d9f07242cf86c073695c63e1770c624f543092c3b4a55f76","observation_id":"8e1b2a61-fe9a-4da8-bcee-5a7041d2c5ce","resolution":{"observed_at":"2026-08-12T15:05:57.856091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.946840Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"99380dc9-ab78-41b2-b335-cd5f9eb81639","year":2019},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.859812Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:116c99fbbee91d4fd96e71fe09577abbc61e2d17da9eb90dfa24cfb3c8a1c896","observation_id":"1ac19980-e07b-4711-8c32-64bc5507a56c","resolution":{"observed_at":"2026-08-12T15:05:57.951457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:05:57.931980Z","title":"Momentum contrast for unsupervised visual representation learning,","venue":null,"work_id":"f1a4242f-b0b7-438c-a728-7e37727b7cd3","year":2020},"citing_paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T15:05:57.863886Z"},"links":{"citing_paper":"/paper/2411.14704"},"observation_digest":"sha256:64d9e78054f22cfa8d92ff24a1fb776074eecbef1840314984a3f966d0930156","observation_id":"852a2f44-d06f-416e-843f-3fb4879354b6","resolution":{"observed_at":"2026-08-12T15:05:57.937860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.14704","last_updated":"2024-11-22T03:28:55Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T14:57:48.429724Z","submitted_at":"2024-11-22T03:28:55Z","title":"Cross-Modal Pre-Aligned Method with Global and Local Information for Remote-Sensing Image and Text Retrieval"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":0,"verified_fuzzy":41},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 1 inbound Pith citation observation for arXiv:2411.14704."}