{"as_of":"2026-08-05T11:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ccb039d8370b423597c7ab597ae4e3c11d7243a323d6506f1d2ea17760da32fa","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T15:37:16.031502Z","state":"measured"},{"denominator":79,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":79,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2604.12159/citation-record","integrity":"/paper/2604.12159/integrity","json":"/paper/2604.12159/citation-record.json","paper":"/paper/2604.12159"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openstreetview-5m: The many roads to global visual geolocation","venue":null,"work_id":"8253507e-2d2e-48da-971f-f57c51ba1689","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:079ff713b7edbeeba4595c038e1d8391c1c39ac44ac906d64bcaacf6014a105c","observation_id":"9821f85b-dd29-44af-85bf-fef6d3b38c67","resolution":{"observed_at":"2026-05-17T19:40:10.234244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:52e1f66b8b8cc3e6a3e4d081692dc8bb61e6c9af086d225a33f8e9223c060c08","observation_id":"1a55a4e6-c8e3-402a-9ae2-9ec7c4c77ed4","resolution":{"observed_at":"2026-05-11T10:11:02.632798Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is space-time attention all you need for video understanding? InICML, page 4","venue":null,"work_id":"38b351a9-5543-4be3-9d8b-a7102a9856c0","year":2021},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:2b31cbd903993b2ae4516bdd18517db181a4af1cf4a11c28773bac3a646e42e3","observation_id":"3fd33036-f77b-49b1-a299-20a85cc9c841","resolution":{"observed_at":"2026-05-17T19:40:10.045260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16423","last_updated":"2025-09-03T02:03:16Z","snapshot_observed_at":"2026-08-04T16:50:51.275310Z","submitted_at":"2025-03-20T17:59:47Z","title":"GAEA: A Geolocation Aware Conversational Assistant","version":3},"cited_work":{"arxiv_id":"2503.16423","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.16423","snapshot_observed_at":"2026-06-28T23:02:46.173454Z","title":"Gaea: A geolocation aware conversational model","venue":null,"work_id":"c1e4cdb3-da9e-41c4-bf94-d3abba3a9c6a","year":2025},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2503.16423","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:905c2abc7ecec2427388378de59628a5e23e29afd6e8fca68a064e984df099ba","observation_id":"da7f6acc-bf63-4078-ae4b-7982894953ae","resolution":{"observed_at":"2026-05-11T10:11:02.494381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Where we are and what we’re looking at: Query based worldwide image geo-localization using hierarchies and scenes","venue":null,"work_id":"33d69dcb-f282-4444-97d1-203e50df1ead","year":2023},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:8e12632df5ef87a0fac840e8871c972282a2c7817b314292f881bbbf01fab2a1","observation_id":"c2122a78-d0bb-40f5-b11a-77cb588ff1e0","resolution":{"observed_at":"2026-05-17T19:40:09.994761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sam- ple4geo: Hard negative sampling for cross-view geo- localisation","venue":null,"work_id":"4bc0e5ad-0fa1-4723-855d-755dbc846cc2","year":2023},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:e044a6898eea8ca8b5a973142c8c159dc21cc9bdc41d218547dd7e5f3044e7b4","observation_id":"ffc09fbc-6f57-4d6e-a140-6e90acb196b1","resolution":{"observed_at":"2026-05-17T19:40:10.135660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":"2010.11929","doi":"10.1175/jcli-d-22-0357.1","metadata_source":"pith","pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","venue":"cs.CV","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","year":2020},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:48bd3a5ebb69ec6670d576cfc9b08fefaae386330f4d08a6e05d19a8c08a84a1","observation_id":"c036f975-cafb-4a28-9499-7f3beb61154a","resolution":{"observed_at":"2026-05-11T10:11:02.603817Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Computing discrete fr´echet distance","venue":null,"work_id":"7c717d25-2d53-4e0a-8f70-39732296f95d","year":1994},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:7c007feaad051fe72da7846cdb98b28009877c11291531730a2d7fb885fb55e4","observation_id":"be4df011-292c-41fb-aa99-18b756724f86","resolution":{"observed_at":"2026-05-17T19:40:10.172233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1902.09516","last_updated":"2019-02-25T18:56:55Z","snapshot_observed_at":"2026-08-04T08:36:21.595551Z","submitted_at":"2019-02-25T18:56:55Z","title":"Condition-Invariant Multi-View Place Recognition","version":1},"cited_work":{"arxiv_id":"1902.09516","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1902.09516","snapshot_observed_at":"2026-07-04T23:25:15.858185Z","title":"Facil, D","venue":null,"work_id":"4f750b47-9bd8-4feb-b955-b4cde652fa24","year":1902},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/1902.09516","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:f89bffb8fd55d8af59c6b76914261a8f08fe0743bdf94d6811503443eb24e8bc","observation_id":"d23cbc21-faab-445a-9afa-89c86d3e583e","resolution":{"observed_at":"2026-07-04T23:25:15.858185Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xi-net: Transformer based seismic waveform reconstructor","venue":null,"work_id":"a2baf605-6030-456c-926d-27318f7e1578","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:1ca5e2c440c649247b2a97d41a82cbaaa35273333b0b4d691c67539edc39c390","observation_id":"55b44c1c-fd3e-4e0e-8aec-8d7030a6bf22","resolution":{"observed_at":"2026-05-17T19:40:10.120313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Garg and M","venue":null,"work_id":"f4c463a0-a30e-4349-91d4-e61dd76cbb44","year":2021},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:61382b9ede3b2d03113b24ba28cf5ac319da46e73b579a776d16127a484f5306","observation_id":"3c10c8e7-8bb4-4281-8f47-2b13ce4005f1","resolution":{"observed_at":"2026-05-17T19:40:10.125978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.00275","last_updated":"2023-02-01T06:44:07Z","snapshot_observed_at":"2026-07-06T14:46:56.842264Z","submitted_at":"2023-02-01T06:44:07Z","title":"Learning Generalized Zero-Shot Learners for Open-Domain Image Geolocalization","version":1},"cited_work":{"arxiv_id":"2302.00275","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2302.00275","snapshot_observed_at":"2026-07-02T21:07:23.782659Z","title":"Learning generalized zero-shot learners for open-domain image ge- olocalization","venue":null,"work_id":"897821cf-56a6-4c61-baa6-ed563d0243eb","year":2023},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2302.00275","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:e20118fa58cdd23cbbe444779c5f53adeff613df9bf0ae4d7e5a669379c25940","observation_id":"56665802-6947-4fac-85f0-93985fa328d8","resolution":{"observed_at":"2026-05-11T10:11:02.576543Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pigeon: Predicting image geolocations","venue":null,"work_id":"1f14f153-aa20-4b32-bec7-6c5dcca1cddf","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:735f066f58569d99f1033d78a34957a0e588695ef7dfd11cddb44c22bb79fbb8","observation_id":"96ca759c-4f90-4cdb-99c3-3d27c88bd6c2","resolution":{"observed_at":"2026-05-17T19:40:10.176059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Im2gps: estimating geo- graphic information from a single image","venue":null,"work_id":"a7654af3-3e1d-46b6-ae23-722d68333a29","year":2008},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:8657e459c0eb72b79b8dd5a490ebb9957d32a91505a8276691fe099ef677a1ba","observation_id":"9a479111-0507-4477-bef5-ea4e1bbd0590","resolution":{"observed_at":"2026-05-17T19:40:10.067130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep residual learning for image recognition","venue":null,"work_id":"bf22cea8-f56f-4f47-9430-0751f0183ab4","year":2016},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:cda96ed68038c46b9442a260ae5fc91839788122f49bb8519cfe31267c3c8d3a","observation_id":"03e5c462-7415-4343-99bc-6759f32571ca","resolution":{"observed_at":"2026-05-17T19:40:10.083335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cvm-net: Cross-view matching network for image- based ground-to-aerial geo-localization","venue":null,"work_id":"8d523ff6-574b-46b7-89cf-36bee215d65b","year":2018},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:91eb39638d2bf090b1d104af7c0533e98d3f9dced111228818748b490bce5f34","observation_id":"55dc9d8d-b76a-48c8-8f65-269082db6759","resolution":{"observed_at":"2026-05-17T19:40:10.102957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"3d convolu- tional neural networks for human action recognition.IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(1):221–231","venue":null,"work_id":"0dd9654c-4483-4438-8459-5f13b1485b7d","year":2013},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:212c61252ba21b377edbb9a7ac6ceb2b33fccdf2c8d73fe8310f8abb1e22693e","observation_id":"5e5972db-d612-4f8f-b392-f5b6e8154560","resolution":{"observed_at":"2026-05-17T19:40:10.142349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"G3: an effective and adaptive framework for worldwide geolocalization using large multi- modality models.Advances in Neural Information Process- ing Systems, 37:53198–53221","venue":null,"work_id":"44486b9c-b196-4e61-9d94-73976ce1290f","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:6e48f6afe73aab3f5212bcaee5d225a8a2941ec7ddd813978182e52feb6d01a3","observation_id":"13bcc413-7249-4c48-a629-1ab9a34f42ff","resolution":{"observed_at":"2026-05-17T19:40:10.138969Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13731","last_updated":"2026-06-24T08:21:04Z","snapshot_observed_at":"2026-07-06T21:26:40.118249Z","submitted_at":"2025-05-19T21:04:46Z","title":"GeoRanker: Distance-Aware Ranking for Worldwide Image Geolocalization","version":4},"cited_work":{"arxiv_id":"2505.13731","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.13731","snapshot_observed_at":"2026-07-02T20:47:22.984867Z","title":"Georanker: Distance-aware ranking for worldwide image geolocalization","venue":"cs.CV","work_id":"8d52808e-a083-45bb-9347-494ba78cb0ba","year":2025},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2505.13731","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:dce032155c62c76d0e6c35cce0bfa9d3333f48f9708e8169366a473614a198b6","observation_id":"6677e735-906a-49ca-92cd-410ee9e88da7","resolution":{"observed_at":"2026-06-25T02:17:29.888408Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":"1412.6980","doi":"10.1002/mrm.28086","metadata_source":"pith","pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Adam: A Method for Stochastic Optimization","venue":"cs.LG","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","year":2014},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:e5cf703c941b5f9f0a7666470edcdcddd5d103908f803136abdb5e5eca62ad0c","observation_id":"163fc47b-a675-4f99-bd49-ea567918cd2b","resolution":{"observed_at":"2026-05-11T10:11:02.460366Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cityguessr: City-level video geo-localization on a global scale","venue":null,"work_id":"10680499-03b7-4b3c-91f0-d3525ac0d165","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:7dccc15e1826f587ce2de1b464b8ba462e82ee647d9991fd33c29a790f971739","observation_id":"3f1962f3-88d3-423f-9f18-eeb2e1b08ab9","resolution":{"observed_at":"2026-05-17T19:40:10.031229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The benchmarking initiative for multimedia evaluation: Mediaeval 2016.IEEE MultiMedia, 24(1):93–96","venue":null,"work_id":"abb025a1-fffc-4852-b8e1-fb2fe5c2b870","year":2016},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:deacf6c5d467a6d74599594607721a47e287b2750377f6c90bbfb2e7d3cb7989","observation_id":"7c511823-3dba-4130-9ad3-3511be810ce7","resolution":{"observed_at":"2026-05-17T19:40:10.089553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Handwritten digit recognition with a back- propagation network.Advances in neural information pro- cessing systems, 2","venue":null,"work_id":"03aba9a0-c651-4488-bf65-7f908930e2a6","year":1989},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:d15ceab282a87a8aa4b5c72be9f2b521f8308fba7f7c9c63ef6fc9598f99b1e1","observation_id":"01274798-9a51-49c0-95f5-5d84a2a1168a","resolution":{"observed_at":"2026-05-17T19:40:10.187334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.13461","last_updated":"2019-10-29T18:01:00Z","snapshot_observed_at":"2026-07-06T08:33:12.534026Z","submitted_at":"2019-10-29T18:01:00Z","title":"BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension","version":1},"cited_work":{"arxiv_id":"1910.13461","doi":"10.48550/arxiv.1910.13461","metadata_source":"pith","pith_arxiv_id":"1910.13461","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension","venue":"cs.CL","work_id":"7ab72623-0d1b-41fc-96e9-181246f2ea00","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/1910.13461","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:9d851941035da7f062fc2fe8c33bfaa9a15b3bf1a62d9ccb5d85c5e57ee7b87c","observation_id":"87b2a349-d954-4524-b4b4-2af50db03fbe","resolution":{"observed_at":"2026-05-13T00:14:58.269455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Retrieval-augmented generation for knowledge-intensive nlp tasks.Advances in neural information processing systems, 33:9459–9474","venue":null,"work_id":"8a908935-30d9-4c8c-a175-126953e5acce","year":2020},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:155e2587d1042dab238e52c1cb38c9b836bd2621cc585a4ca01fcd46cc71ad93","observation_id":"4a8e8144-1f57-434e-8006-29e500f043c0","resolution":{"observed_at":"2026-05-17T19:40:10.168594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Georea- soner: Geo-localization with reasoning in street views using a large vision-language model","venue":null,"work_id":"2d4d3ff5-6af7-4481-bb63-2499f96a444e","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:8020c90e2db9db78aa8ee24041c92ff9f026f6c32fa1cf36bb9491ee146848e0","observation_id":"b03ac706-e594-452d-b4c5-3eb4b6ee1af7","resolution":{"observed_at":"2026-05-17T19:40:10.231030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lending orientation to neural networks for cross-view geo-localization","venue":null,"work_id":"9b1c2c37-b56a-4a9f-8385-1159c8ed7ce5","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:187c3241fe8444add4fbd4a32f1b7dd3c730b99abf84d7222c53e5de13769157","observation_id":"2acd3245-4478-447e-b102-5320ba74841c","resolution":{"observed_at":"2026-05-17T19:40:10.054742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Swin transformer: Hierarchical vision transformer using shifted windows","venue":null,"work_id":"4e272cd8-b145-4681-b239-16274c8edabc","year":2021},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:ccd83c2cf7021c9d1ea99435b4afede35757726cd21b1bb9e3fc4ff9ad2c1863","observation_id":"8897277d-ba04-4959-add9-006dd37008d8","resolution":{"observed_at":"2026-05-17T19:40:10.079906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mereu et al","venue":null,"work_id":"9d02b765-d324-4a19-9810-662391c50a7d","year":2022},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:9b7498194f66de4539fe314915e3932a81a6e5d490aa57873b34137e43cb3f4b","observation_id":"b6855385-8d9a-4a44-9768-45d2ad88efde","resolution":{"observed_at":"2026-05-17T19:40:10.092885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13965","last_updated":"2024-09-04T18:32:38Z","snapshot_observed_at":"2026-07-06T17:47:59.544888Z","submitted_at":"2024-03-20T20:37:13Z","title":"ConGeo: Robust Cross-view Geo-localization across Ground View Variations","version":2},"cited_work":{"arxiv_id":"2403.13965","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.13965","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Congeo: Robust cross-view geo-localization across ground view vari- ations","venue":null,"work_id":"c920e93e-2f5b-4334-8472-725a3af2dbe8","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2403.13965","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:f9cd7a00324b692c373c92c9b9a628be134bea69d6c8682507efc66788607927","observation_id":"861f9c45-f3e1-4a91-a776-29211e3b5701","resolution":{"observed_at":"2026-05-11T10:11:02.482280Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08681","last_updated":"2020-08-13T05:42:12Z","snapshot_observed_at":"2026-07-06T08:16:16.271286Z","submitted_at":"2019-08-23T06:22:06Z","title":"Mish: A Self Regularized Non-Monotonic Activation Function","version":3},"cited_work":{"arxiv_id":"1908.08681","doi":null,"metadata_source":"pith","pith_arxiv_id":"1908.08681","snapshot_observed_at":"2026-07-08T11:24:54.882129Z","title":"Mish: A self regularized non-monotonic activation function","venue":"cs.LG","work_id":"f314c3b6-cec7-4246-8006-cbed6b9840d3","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/1908.08681","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:493281d48eff5d16b677618f0f9f4922bde3b0954e342e329f44a4718a70a6da","observation_id":"37c45461-28c3-4851-a987-984b78ff0629","resolution":{"observed_at":"2026-05-11T10:11:02.509102Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Geolocation estimation of photos using a hierarchical model and scene classification","venue":null,"work_id":"90fb08a2-c862-4a72-bd26-b6e9b52b7485","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:b33bd5faf347e5a9a8a61fa724178070d209244ac26fcd7d5512897494bb1384","observation_id":"f34e7bd2-823e-40ec-82d0-0616c8c4bc54","resolution":{"observed_at":"2026-05-17T19:40:10.183642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":"2304.07193","doi":"10.48550/arxiv.2304.07193","metadata_source":"pith","pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DINOv2: Learning Robust Visual Features without Supervision","venue":"cs.CV","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","year":2023},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:c16fa72ca18b96a78b7deaab8b61b0bb4b7dd2711481f1c8086bfa07a9884779","observation_id":"24b1f072-89c6-4f16-8633-b5359be3dae7","resolution":{"observed_at":"2026-05-11T10:11:02.533673Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pytorch: An im- perative style, high-performance deep learning library.Ad- vances in neural information processing systems, 32","venue":null,"work_id":"753eea04-5547-41a1-a823-bd2ceb4898ae","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:304673cab8642ebaa616175f7da6cbea0405ff52d04c58f1b24902af5888b9f6","observation_id":"b3b85a14-6bae-47a5-bf93-99ce3807da00","resolution":{"observed_at":"2026-05-17T19:40:10.086531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4f860ba0-03b1-4627-9638-987870c3d11a","year":1901},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:5bd4b66f003b8d6f4f582d62d0135ccdd91cb4da327f34b27f2230e733ae6b38","observation_id":"00a87966-ab28-4326-8f84-b13c0990575b","resolution":{"observed_at":"2026-05-17T19:40:10.222582Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.02840","last_updated":"2024-08-05T21:29:33Z","snapshot_observed_at":"2026-07-06T18:57:14.251413Z","submitted_at":"2024-08-05T21:29:33Z","title":"GAReT: Cross-view Video Geolocalization with Adapters and Auto-Regressive Transformers","version":1},"cited_work":{"arxiv_id":"2408.02840","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.02840","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Garet: Cross-view video geolocalization with adapters and auto-regressive transformers","venue":null,"work_id":"2291643e-83ce-48b4-9d05-61fa64dbda90","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2408.02840","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:6fc0683cdc077857cee0fb9d10be90e77aa973fb1cdac7d0fea5a57ade0643ff","observation_id":"cad4028c-8d36-4655-8526-8593bc8c7ead","resolution":{"observed_at":"2026-05-11T10:11:02.554075Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Where in the world is this image? transformer-based geo-localization in the wild","venue":null,"work_id":"3cda1d20-96d6-4bd1-8f77-19a665839aa9","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:937e5e39b740e9c85ab22c1da3092a8f32d1ae45392e95c98fc0119392e3c656","observation_id":"4007da5c-8d6f-4e04-971f-3c55f3d308b6","resolution":{"observed_at":"2026-05-17T19:40:10.109949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6ead74d5-d71f-4494-86da-2635b84036c3","year":2022},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:6d7467b591bae0656498c2dcbad9b4048a98bd28bd408b9206f9500c63a7dcdc","observation_id":"15d0c818-5d3a-4283-8b34-a77ef687ffdd","resolution":{"observed_at":"2026-05-17T19:40:10.096338Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models are unsu- pervised multitask learners.OpenAI blog, 1(8):9","venue":null,"work_id":"68a06a92-e705-4d4c-b7aa-6600434c4a94","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:5af257241ee527af2d44d865fac6e9a1955e5897df09d5c2e25655c99ee350a4","observation_id":"a8a7f8a7-6a94-43a8-835c-a7daf8529522","resolution":{"observed_at":"2026-05-17T19:40:10.001694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":"90735b2c-f5fa-4f54-a397-fe0357059c78","year":2021},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:0d182d5a2071ba44c169c1f494b04fac5563ea16ec73913209e96aa51cc588cd","observation_id":"5e9c9d57-4ce9-447c-a5b6-e22e3525e226","resolution":{"observed_at":"2026-05-17T19:40:10.035218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cross-view image synthesis using geometry-guided conditional gans.Computer Vision and Image Understanding, 187:102788","venue":null,"work_id":"e2a58228-7efc-4c56-b261-fb0bf98c4a2e","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:5dc2443e6fcf2f43463a0148149f1da5750eadf163c057ba0a0a9814cdf1a40f","observation_id":"6346e42f-a180-4f8e-8bd5-1223262c1247","resolution":{"observed_at":"2026-05-17T19:40:10.021116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bridging the domain gap for ground-to-aerial image matching","venue":null,"work_id":"a1084251-7801-45c4-906d-51cac8ad0767","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:0c106617d7cfc0c5489e61c40734675208b35147cf3aa34ab8597f14a0dc20b4","observation_id":"ecbc984a-5752-4bba-875d-53922bbbd48b","resolution":{"observed_at":"2026-05-17T19:40:10.129142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video geo-localization employing geo-temporal feature learning and gps trajectory smoothing","venue":null,"work_id":"35ec9132-71c4-40ea-9213-5bd3bfe99eb0","year":2021},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:02ceb2669384f5d30d6b267e66bf4b7d17f12b1d42a155bf8c63f81c840382a7","observation_id":"c954a814-41de-47c3-aa98-1ba7a6adc2ce","resolution":{"observed_at":"2026-05-17T19:40:10.152554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generalized in- tersection over union: A metric and a loss for bounding box regression","venue":null,"work_id":"5350b54b-7b71-4dad-84ee-f1a8a2839e9c","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:865352cebee1501b9608359ab021291d296d0e09968224ea5db2e59fd5e946d9","observation_id":"60b8ce06-2bd4-40ac-90f0-b0bbbf2f6c16","resolution":{"observed_at":"2026-05-17T19:40:10.038925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The equal earth map projection.International Journal of Geographical Information Science, 33(3):454–465","venue":null,"work_id":"8180dba9-89e9-480d-9195-232818b37130","year":2019},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:8e4d856d3bc6d1b3cfc6385d10b9084c7d42413fb602c775c88697157f4b6c3d","observation_id":"a83425b5-1d6f-415a-aa33-05099ddd9db5","resolution":{"observed_at":"2026-05-17T19:40:10.106328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cplanet: Enhancing image geolocalization by combi- natorial partitioning of maps","venue":null,"work_id":"ce327d60-c9ef-41bd-b637-b966a25e030d","year":2018},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:08f09ec3bba239ceb2fcbdda514069b6e4d4fd2a224bda989de09e768846a994","observation_id":"7609aa96-2979-4323-96e7-1e6cda7e7f9f","resolution":{"observed_at":"2026-05-17T19:40:10.064010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gt-loc: Unifying when and where in images through a joint embedding space","venue":null,"work_id":"2b345b2b-8f17-4143-8b0f-becd3e04da04","year":2025},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:31f8acd7a2ab4b305dc7322b900aa46eb8743fbe7071d3e4f7243dc9d12ff577","observation_id":"9028f369-62f3-49d5-9da8-c3a6d1d582df","resolution":{"observed_at":"2026-05-17T19:40:10.042196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Where am i looking at? joint location and orientation es- timation by cross-view matching","venue":null,"work_id":"5531a81d-9944-449d-90bc-96c3b963da89","year":2020},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:d0778ff64185acf86a3356bbd4d47174bdb5932610fec05434a6cb869d86c289","observation_id":"823c50c6-4260-4fd3-9e01-908a4c89ef17","resolution":{"observed_at":"2026-05-17T19:40:10.214020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1409.1556","last_updated":"2015-04-10T16:25:04Z","snapshot_observed_at":"2026-07-06T03:53:32.549552Z","submitted_at":"2014-09-04T19:48:04Z","title":"Very Deep Convolutional Networks for Large-Scale Image Recognition","version":6},"cited_work":{"arxiv_id":"1409.1556","doi":"10.48550/arxiv.1409.1556","metadata_source":"pith","pith_arxiv_id":"1409.1556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Very Deep Convolutional Networks for Large-Scale Image Recognition","venue":"cs.CV","work_id":"1c4b4409-c14b-488b-a086-c57a5aab8a29","year":2014},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/1409.1556","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:0bc2d1ad40f54ff1bc04bf3746c9169012cd5406b196c671e750ac74865a1d84","observation_id":"9ca934de-886f-46a9-a3db-e6b8b9db1d1f","resolution":{"observed_at":"2026-05-11T10:11:02.623353Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-12T23:51:07.915518+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T23:51:07.915518+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fourier features let networks learn high frequency functions in low dimen- 10 sional domains.Advances in neural information processing systems, 33:7537–7547","venue":null,"work_id":"7f8f4730-c4e3-4791-a540-5ca6f2fbfd22","year":2020},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:e23b8e8c4229484231e3530e673d6d15a637e62681c141f934e164fd2cf9f6b6","observation_id":"06d2f80a-f05e-4ef0-96b6-9c9916363398","resolution":{"observed_at":"2026-05-17T19:40:10.074386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Coming down to earth: Satellite-to-street view synthesis for geo-localization","venue":null,"work_id":"d3c5a71f-6a3b-4659-a00e-c4ff8f663f9b","year":2021},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:4703388937ffc779bb82aa81b8ff1f32df9609e66672543d160c061ebfa2e248","observation_id":"2d558be8-832b-4f80-992e-37ecacc0240a","resolution":{"observed_at":"2026-05-17T19:40:10.218699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training.Advances in neural information processing systems, 35:10078–10093","venue":null,"work_id":"60f21ee0-4081-43e2-8822-e36af0b979a6","year":2022},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:af2d3a189ce83a7eff2a14b4e2b2843349ddf738feca763c81c292b49e09ff1b","observation_id":"8eee83f0-6d0f-486a-aab2-153a7765bd65","resolution":{"observed_at":"2026-05-17T19:40:10.099804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"City scale geo-spatial trajectory estimation of a mov- ing camera","venue":null,"work_id":"70c4a754-7050-4441-a1d3-7da0d602244c","year":2012},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:bb1c3a91bd6a849e66eb42f9268833a6727dba00e6d0e78fb4156cfc984c7eea","observation_id":"b27accaf-d40c-48e5-8fa2-3f87ccdcb63e","resolution":{"observed_at":"2026-05-17T19:40:10.180087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Attention is all you need.Advances in neural information processing systems, 30","venue":null,"work_id":"e0e61761-325e-4a2d-9a9f-9df31b66da62","year":2017},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:4184418e572fc9735a548d1471c596fd7630b3934e78979cdc5f49e302e85258","observation_id":"a69a32f8-0f82-4aec-bbb6-37417e906c5b","resolution":{"observed_at":"2026-05-17T19:40:10.191044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Geoclip: Clip-inspired alignment be- tween locations and images for effective worldwide geo- localization.Advances in Neural Information Processing Systems, 36","venue":null,"work_id":"8c0b5f13-f25d-4051-82e5-11916b12c226","year":2024},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:098e7897cd7b80fd2a63e399742c973449956e8ef9285d579ed86a2af82a93f4","observation_id":"4fe7ce10-1093-4f8f-bcb4-4bab0f93d6d0","resolution":{"observed_at":"2026-05-17T19:40:10.113453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Revisiting im2gps in the deep learning era","venue":null,"work_id":"9cd61462-021d-4aea-bb74-c3a53287d6bc","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:ed7de8c9458b9794c203f8cc411609d7e39215090bc945cb227b95848fb48af4","observation_id":"d9891760-cd4f-4c25-b0f1-cb89b3e8581b","resolution":{"observed_at":"2026-05-17T19:40:10.017112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gama: Cross- view video geo-localization","venue":null,"work_id":"7536f112-b4ed-4fc1-a6dc-440311a1393b","year":2022},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:5026083b43fcee56e3301e3682af17c8d1f0fa77b2790ff999fc28c97822f595","observation_id":"48c0d0e8-887d-45b8-b7a4-da094bfa119f","resolution":{"observed_at":"2026-05-17T19:40:10.149366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.05171","last_updated":"2022-06-14T14:16:52Z","snapshot_observed_at":"2026-08-03T16:40:41.812585Z","submitted_at":"2020-10-11T05:36:54Z","title":"fairseq S2T: Fast Speech-to-Text Modeling with fairseq","version":2},"cited_work":{"arxiv_id":"2010.05171","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2010.05171","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fairseq s2t: Fast speech-to-text modeling with fairseq","venue":null,"work_id":"3afd16f7-84c9-4962-bfa3-a78fcb06f537","year":2010},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"cited_paper":"/paper/2010.05171","citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:ca2f7b00f771b8e13208249aa76699384cc00c2bfc68832ed7fce696f19022ec","observation_id":"b6d3b66e-279a-44ea-ab44-187aabe4edd5","resolution":{"observed_at":"2026-05-11T10:11:02.475816Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mapillary street-level sequences: A dataset for lifelong place recognition","venue":null,"work_id":"80e5404a-8a60-4d7e-93d7-4aa8325aad2f","year":2020},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:6f9cc03846f75f0434113247414a852ce64c75130a28899d6e46890beece9006","observation_id":"2c1ae000-a86e-4095-a839-ba23421e9284","resolution":{"observed_at":"2026-05-17T19:40:10.145808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Planet- photo geolocation with convolutional neural networks","venue":null,"work_id":"2ba18e84-2ef7-4e24-8bdc-9b31ca1e33c7","year":2016},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:9cdb957c0c658987c0b4d01c07819f48910db7173eccec24461eb18d0e7db56f","observation_id":"1f7fb1da-ddf5-405a-9384-836c28c90bb7","resolution":{"observed_at":"2026-05-17T19:40:09.986691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wide-area image geolocalization with aerial reference im- agery","venue":null,"work_id":"f7e344f7-f687-4a32-9142-9351d25e0b5d","year":2015},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:ea068c23f6f5014a3b7e17a92a2d66f90e1f8426249fae5bc045b2dafa8b2d13","observation_id":"c0ef8d27-5505-4395-bbfc-7781fa5e8317","resolution":{"observed_at":"2026-05-17T19:40:10.226812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cross-view geo-localization with layer-to-layer transformer.Advances in Neural Information Processing Systems, 34:29009–29020","venue":null,"work_id":"ae639766-1fb7-4634-a2c4-69f9c16c6417","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:979fc21b8780ddef603f21968add0c28318fb37c625f038b7853a4a56139c097","observation_id":"cafbd990-faf6-46d0-be3d-7b42a12adb31","resolution":{"observed_at":"2026-05-17T19:40:10.117156Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bdd100k: A diverse driving dataset for heterogeneous multitask learning","venue":null,"work_id":"aef641cc-7792-4db3-94bc-27fc479fb08b","year":2020},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:99feb603b74c11340316b11674779f38282c28957dc025d91223b4748f5cde43","observation_id":"b93c00ac-06f6-40fb-b1b1-fc65248d2931","resolution":{"observed_at":"2026-05-17T19:40:10.123217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"9b9716f4-cc75-40f8-8f9b-26ea427b73f8","year":2023},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:13be641184d2335fdc9dfbdfddbb57dc87ecc15da16bb9daf42e8449cf1cfd3e","observation_id":"a21ee39c-aa93-4b00-ab25-1d476df4c22f","resolution":{"observed_at":"2026-05-17T19:40:09.991174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Places: A 10 million image database for scene recognition.IEEE transactions on pattern analysis and machine intelligence, 40(6):1452–1464","venue":null,"work_id":"5ad05f19-bf71-4d65-bd85-19311e006573","year":2017},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:a128c0d34998322d49eda065941fb6e1ee391478fe3a3c506ea3ef60c96a3d56","observation_id":"7a8910d8-cbdc-4c87-a9d2-d677adfc32b5","resolution":{"observed_at":"2026-05-17T19:40:10.025548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vigor: Cross- view image geo-localization beyond one-to-one retrieval","venue":null,"work_id":"ad84ed5c-c51b-4b15-93bb-e767d7943420","year":2021},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:89d6c0faea8b2ce1f12cdc146cebc9a7416a43987977943ab0f0444d2ac7c8ec","observation_id":"0bb0feb4-d82f-456f-92de-d1488567abda","resolution":{"observed_at":"2026-05-17T19:40:10.132450Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transgeo: Trans- former is all you need for cross-view image geo-localization","venue":null,"work_id":"6a83ce6d-7788-4650-87e6-902d04c953e5","year":2022},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:33478f23ae64d1b7b97d082e966df08cc92629456e4771781a2b2db450dc84b6","observation_id":"cbc85809-bd1f-42a7-a566-30467dc98112","resolution":{"observed_at":"2026-05-17T19:40:10.012305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"(If there are a few outliers they can be skipped)","venue":null,"work_id":"2d5582cf-6a81-4df1-81fb-cb778c325f9d","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:9a05749855bc2ab2f29859adb7a05c17f2c1f7afe62e28ba8a66125d1154e54d","observation_id":"92610c0a-e5e4-4013-a3b9-7c1106e3914d","resolution":{"observed_at":"2026-05-17T19:40:10.155922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Also determine the resolution of the gallery (the finer the resolution the larger the gallery","venue":null,"work_id":"2a438689-8135-4e17-91c6-d8c4c03780c9","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:5250d80601e08b9d1a9ea2b1b6c89da4207a7af91dc4efd89c4c80329b741e35","observation_id":"d199c9e3-30a7-4208-804d-29887df72c1e","resolution":{"observed_at":"2026-05-17T19:40:10.061056Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0cd76c52-d119-4c63-b3fa-5cca828b6a8f","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:ca2e439a058c4dff1fb040b399951bfc579b3b97bc64284a828472db7344b13b","observation_id":"9cbe8649-66c7-43f7-a667-ea0f06c67315","resolution":{"observed_at":"2026-05-17T19:40:10.163421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d6aa869b-adc2-426b-9530-708b074fdde0","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:fc39951e879537315215d7e2503b4debb1e0f5aebea2a2b51a36835681ba7126","observation_id":"c435deef-c37f-4bf1-86d3-45a90f1bd3f6","resolution":{"observed_at":"2026-05-17T19:40:10.051430Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"205f32ba-43ef-4c9d-a1e1-f7d8b5af7c04","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:26790e5ab05c04cbb078181f892826978f051bad2baf81f2d2673ae3ec2c2c6d","observation_id":"195641ed-dd49-40fc-b644-3b5ea8d5dc7e","resolution":{"observed_at":"2026-05-17T19:40:10.048329Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d353fdd2-490c-418c-ae93-f5ab5897edbc","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:386a9b4fead95391a94e52640640d526a1e875963c081c9630c589c28e7ad413","observation_id":"136c5760-b6e2-4eb2-bf6e-119bbc178dde","resolution":{"observed_at":"2026-05-17T19:40:10.057871Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This will serve as the ground truth label of the entire video sequence","venue":null,"work_id":"37554629-fd21-4098-b129-e96b38acd5b6","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:55e2e8154d973acdf6262faebc827fcd37033825f7e8eb52c5be06dc541faf6d","observation_id":"0cad1e0b-cc1a-40c9-8486-34ea4535f529","resolution":{"observed_at":"2026-05-17T19:40:10.008356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Obtain the prediction that is closest to this centroid (this reduces error due to out- liers)","venue":null,"work_id":"62d18093-eb3f-45e0-8ede-8324c0fe1074","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:6088011bfde8dd47aa24dc351c93004995d134975677913a8a8c94c6c3c19cbd","observation_id":"15fbe9a9-3b93-4cd1-90d2-6b31552ebd1d","resolution":{"observed_at":"2026-05-17T19:40:10.159738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Compute distance accuracy at all thresholds using this distance measure","venue":null,"work_id":"43060e1f-8144-40c6-81d4-fc9ce18e93bf","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:e32a088117634419924141d6a6628a1637924a10bea84bdf4f9a7a2b6a5a1cad","observation_id":"8a51bb11-ed4c-4ee0-87ab-147c1f016444","resolution":{"observed_at":"2026-05-17T19:40:10.070759Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1f1ae2c1-48b8-4556-acff-ab85b0167e5f","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:5948f2b672270c2e0b47caa0d1512d83554df9e3833045e25edcccd893d57329","observation_id":"ba5aa688-d127-4f94-847f-84ac4469a9fd","resolution":{"observed_at":"2026-05-17T19:40:10.077000Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"(b) As CityGuessr68k is very large, we sample the data in such a way that the number of sequences is roughly equivalent to MSLS","venue":null,"work_id":"c7b7b88c-a2b2-4ac0-8ca8-ea5838629faf","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:9cef2dacbb7186a151d473ec0774be72cebf53107448924adccee466db9982c7","observation_id":"223f54a2-361b-4743-8715-79ccff5c4253","resolution":{"observed_at":"2026-05-17T19:40:09.998205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We train a model on this unified data for 200 epochs at an learning rate decay rate of 0.97","venue":null,"work_id":"fa4ed501-e91c-4cb7-a555-081687b65790","year":null},"citing_paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-10T15:37:16.031502Z"},"links":{"citing_paper":"/paper/2604.12159"},"observation_digest":"sha256:082b9615f68c9b20d2de20970535ca1b9e00093998ef013e23442ca7ccca5d78","observation_id":"ba8e19a2-9419-4371-806c-7e885756aa3b","resolution":{"observed_at":"2026-05-17T19:40:10.004948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.12159","last_updated":"2026-04-14T00:35:07Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-14T00:35:07Z","title":"VidTAG: Temporally Aligned Video to GPS Geolocalization with Denoising Sequence Prediction at a Global Scale"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":6,"verified_exact":14,"verified_fuzzy":57},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 0 inbound Pith citation observations for arXiv:2604.12159."}