{"as_of":"2026-08-18T10:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1c2f3ea23f9a82852e523516d7e6c89d4e9aa2a93f2a0b94aeddca3929dae4fe","coverage":[{"denominator":69,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T00:42:54.247849Z","state":"measured"},{"denominator":69,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":69,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.19412/citation-record","integrity":"/paper/2412.19412/integrity","json":"/paper/2412.19412/citation-record.json","paper":"/paper/2412.19412"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.985235Z","title":"Thermal voyager: A com- parative study of rgb and thermal cameras for night-time au- tonomous navigation","venue":null,"work_id":"fe0aac53-2d50-487f-bb7e-ce86d03dfc6b","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.855805Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:f060676e5b546fb392f7f9c28600c4ef0e4ffbc83f2dbb7a6bab3cac513945ca","observation_id":"f17fb963-42eb-4f74-84ab-9c4ad2964d2d","resolution":{"observed_at":"2026-08-11T00:42:54.988870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.975101Z","title":"A low power, fully event-based gesture recognition system","venue":null,"work_id":"10759363-5a98-43f6-9257-4f78c5a4a8f6","year":2017},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.860418Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:d78b436f67ddf799f6de09e786432a1b19c166797d603ff686a8be0a074286b8","observation_id":"38248371-4b93-47b3-98e8-009bfa8b2082","resolution":{"observed_at":"2026-08-11T00:42:54.978680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:53.864050Z","title":"Netvlad: Cnn architecture for weakly supervised place recognition","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.864050Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:3009612af2aeabdf0ee39ec403582c48be660f7cb90b074b3e01a0ade158c6e5","observation_id":"e8b8ef58-ecdf-4d0d-9433-dbd158ed2542","resolution":{"observed_at":"2026-08-11T00:42:53.864050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.958990Z","title":null,"venue":null,"work_id":"3b7e87ec-90fc-4ba9-95f6-6dc29b58462e","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.867596Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:86c974d98d8821559360eed79ff308f513d2f28324a254864bf55df9e348c14e","observation_id":"717d221d-1cf7-44b8-8809-086cfe8cdb24","resolution":{"observed_at":"2026-08-11T00:42:54.962545Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.949193Z","title":"Learning to match features with seeded graph matching network","venue":null,"work_id":"e5ab3bb5-076c-4e69-a277-a84e2206d306","year":2021},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.871204Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:84f3f3a2ade2fd8314874fdc72db6a4cab7c3ccd79fbc75f6c170932419bc07a","observation_id":"59c05108-7b84-49c5-9d57-5131e11cd6cf","resolution":{"observed_at":"2026-08-11T00:42:54.952644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.937719Z","title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes","venue":null,"work_id":"c369efa2-c1ca-4e9f-b494-ff9e76a9c269","year":2017},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.875105Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:5a622c6333f681685214bfcbf32f39f7aba57912a6e4e23230b9cfa6529f46bd","observation_id":"896ef7cb-7c12-4d86-94e5-3fd3868c4d56","resolution":{"observed_at":"2026-08-11T00:42:54.942110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.926593Z","title":"Crosshomo: Cross-modality and cross- resolution homography estimation","venue":null,"work_id":"b3b79035-a9f6-461c-9ce7-bdf463a55391","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.879304Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:33eb27978c724e3c51f02151a3ae257a9b9c75ffc6c2c9e9a8ea82c4fbffc959","observation_id":"077f06d1-1b7b-4f63-85c4-1bbd3b0fe72c","resolution":{"observed_at":"2026-08-11T00:42:54.930460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.916249Z","title":"Redfeat: Recoupling detection and description for multimodal feature learning.IEEE Trans","venue":null,"work_id":"042d68aa-8f18-4d8a-8715-2e38beda8710","year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.883338Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:ccfb5aee0fc4f321c6e279730b4afa6c55ab69d5ead08fc2628fac895fbab1a1","observation_id":"322ac734-74d7-4d42-a8fc-70e9f0a8c369","resolution":{"observed_at":"2026-08-11T00:42:54.919940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.905016Z","title":"Superpoint: Self-supervised interest point detection and description","venue":null,"work_id":"4344e6ee-a065-403d-9c9a-aa4782ca4434","year":2018},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.886746Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:ad3ec00acbf44592fe16afec13a74f5084da48f6147c81a848daf565a8c5a9b3","observation_id":"fd4f2fff-22ec-4deb-b0b7-3453bdb91a86","resolution":{"observed_at":"2026-08-11T00:42:54.909111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.893548Z","title":"Dkm: Dense kernelized feature matching for geometry estimation","venue":null,"work_id":"af1317ee-3f76-4673-a2d7-0753a3f4eaf1","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.890250Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:7b6d3bb3353ea48f08e0e1652d89e0409cde0398469162b163a9b36cb555e5a4","observation_id":"a57eb552-0db6-43a8-9594-4da525c98ea4","resolution":{"observed_at":"2026-08-11T00:42:54.897801Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.882647Z","title":"Roma: Robust dense feature matching","venue":null,"work_id":"a54fe663-77d7-448b-b398-74fed94fba44","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.893855Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:da96154bc452e48ed6302a68bef76eb3f326df664f09ab070c3728290dd37a9b","observation_id":"486e7d23-6be3-4463-82c2-527f398fca43","resolution":{"observed_at":"2026-08-11T00:42:54.886652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:53.897466Z","title":"Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography.Communications of the ACM, 24(6):381–395, 1981","venue":null,"work_id":null,"year":1981},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.897466Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:a8e60fc05cf632bea777687be32e0c9595c46cf3d080f0b36b52240bb24ae2bb","observation_id":"9695bc5a-1e86-4beb-a572-144e054d5723","resolution":{"observed_at":"2026-08-11T00:42:53.897466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.865945Z","title":"Event-based vision: A survey","venue":null,"work_id":"59755cf0-51e1-4e46-b606-053ea4c81638","year":2020},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.901145Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:ef4af34da779b99da380f18ec9d15f6e02fe76791ee2a717c730fd1c218571ca","observation_id":"c03dc338-d09c-45b9-9dfb-bbd603670e80","resolution":{"observed_at":"2026-08-11T00:42:54.869736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:53.904538Z","title":"Low-latency auto- motive vision with event cameras","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.904538Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:424e2e43bfc505ba699cfd5d0ecdab840034647f5110d9e01157eed946af4f6e","observation_id":"2e2e35bb-a442-4e55-b1ae-60652e60e6da","resolution":{"observed_at":"2026-08-11T00:42:53.904538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.847763Z","title":"Video to events: Recycling video datasets for event cameras","venue":null,"work_id":"9bd82e60-9802-41af-aa31-bc53488aec48","year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.907975Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:cd85c89b133f8037eb6b21904285ef7efb0a85d1ab9332e43c2999a24b52d8e9","observation_id":"7897b29d-204a-4aae-9bf5-2878ed0bc22c","resolution":{"observed_at":"2026-08-11T00:42:54.851662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12154","last_updated":"2024-12-15T15:31:56Z","snapshot_observed_at":"2026-08-16T13:59:46.348617Z","submitted_at":"2024-04-18T12:58:55Z","title":"StyleBooth: Image Style Editing with Multimodal Instruction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12154","snapshot_observed_at":"2026-08-11T00:42:53.912203Z","title":"Stylebooth: Image style editing with mul- timodal instruction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.912203Z"},"links":{"cited_paper":"/paper/2404.12154","citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:b4d7714adce3de2cbef2cfa34ddb508d0357fba67a7ed904ecf38e67941e70b6","observation_id":"b5ce5036-a1e8-409c-8c1a-db8bb8fc2b19","resolution":{"observed_at":"2026-08-11T00:42:53.912203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:53.916125Z","title":"Denoising dif- fusion probabilistic models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.916125Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:abfbc7e707a63ca8a1804a9d545f4a4a242775b3f03f99429a6d7765c48e3c57","observation_id":"9818c6c9-b3cb-4ba4-aa62-3ea17ce75896","resolution":{"observed_at":"2026-08-11T00:42:53.916125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.829889Z","title":"Pos-gift: A geomet- ric and intensity-invariant feature transformation for multi- modal images","venue":null,"work_id":"498b7c33-2571-45a0-929c-e6281b755667","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.920022Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:1dcc2bd263f0ce47537471707aae91b07cea97c7a31349ffe2f4af328e98d2f6","observation_id":"920e4324-70ca-4144-ace2-9d6091c2447f","resolution":{"observed_at":"2026-08-11T00:42:54.833788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:53.923553Z","title":"Lora: Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.923553Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:93674dff6e7cc3b8a098436b7fb9c44fae9e73728a28e43c78399591c5501887","observation_id":"dfa6f7db-8bb2-4e75-b43c-c68a4c8aef7d","resolution":{"observed_at":"2026-08-11T00:42:53.923553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.813607Z","title":"Llvip: A visible-infrared paired dataset for low-light vision","venue":null,"work_id":"363925ef-4209-47cd-a35d-539d30f11b19","year":2021},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.927101Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:4f2dcf31355de5575701e2fbe7be15e7c489f2125b95c2ebec6c1e852499a5cc","observation_id":"f8a86ae3-eec6-494d-a208-ec9820ec66d5","resolution":{"observed_at":"2026-08-11T00:42:54.817228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.803086Z","title":"Omniglue: Generalizable feature match- ing with foundation model guidance","venue":null,"work_id":"fd94ca14-f01e-414d-a709-076445652d8f","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.931385Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:3de0f17828df41b1d6a7c8f2f068692022eff855f7a119fc662e5773fb6802b2","observation_id":"16b9046b-249f-4335-8a80-391f2ad5209a","resolution":{"observed_at":"2026-08-11T00:42:54.806598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.792285Z","title":"A review of multimodal image matching: Methods and applications","venue":null,"work_id":"e5a3bd18-8dd0-4e48-bd31-62ac22a91411","year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.935268Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:7314c2535907bc63374ac4db77bc9de66057113119aa5d09e59802a9cb24e6d2","observation_id":"c137b20f-2267-4a03-b088-c97d1695bc36","resolution":{"observed_at":"2026-08-11T00:42:54.795901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.772095Z","title":"Imagenet classification with deep convolutional neural net- works","venue":null,"work_id":"3d8b34c0-6159-4400-bd67-37f2a2c92a47","year":2012},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.942684Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:097d1f080f5cd298987327cb05fd6e53a1b8c77afad698a907da7fa2039bd870","observation_id":"387c5b74-7ffc-4976-ab1a-8ed65b71305a","resolution":{"observed_at":"2026-08-11T00:42:54.775660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.761333Z","title":"Rift: Multi-modal image matching based on radiation-variation insensitive fea- ture transform","venue":null,"work_id":"ad4e9b57-59f0-4889-a09a-6d02b60962dd","year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.946390Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:95ea950ee8cb50f8822e51952494678760d4f34725e163d7d32fa38799df1e55","observation_id":"192747da-71cc-424e-98db-9880bb7195ac","resolution":{"observed_at":"2026-08-11T00:42:54.765523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.750314Z","title":"Lnift: Locally normalized image for rotation invariant multimodal feature matching","venue":null,"work_id":"e8e26a07-a725-4469-8a68-3f60d6c52a94","year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.950211Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:734213ba222a9eee252d75e6bf9d36d371e9d90cb352a2a6dfc3d8f498166e81","observation_id":"f3497f4f-5d97-464c-b8dd-e82557e0637e","resolution":{"observed_at":"2026-08-11T00:42:54.753997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.740008Z","title":"Multimodal image matching: A scale-invariant algorithm and an open dataset","venue":null,"work_id":"7bc81b93-5d43-443d-a19b-6ac56325efb4","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.953653Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:3e0cb89a9ae8ce9bcaf7684742b0e4246079b9784183a85e2bf232628d3abd9e","observation_id":"22074aa3-cd5b-40db-a8f9-2a8cf39e746b","resolution":{"observed_at":"2026-08-11T00:42:54.743688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.727672Z","title":"Megadepth: Learning single- view depth prediction from internet photos","venue":null,"work_id":"c8b568f7-1339-44cd-8a10-633b6280f119","year":2018},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.957458Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:796846256a2e1282af12db45791f798f56dcc9a30f4d13cf835ef38e5bc25a83","observation_id":"ba2cb257-c6ae-42c1-8d33-0ada247e41e5","resolution":{"observed_at":"2026-08-11T00:42:54.732122Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.715634Z","title":"A 128 × 128 120db 15µs latency asynchronous temporal con- trast vision sensor","venue":null,"work_id":"beb9eda8-691b-455f-9890-eda5f1fcab0a","year":2008},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.960814Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:c49e1d4faf8884eb8a62f27f053770eb4f4bd0bcaf9fb0bb85475a66299f80a6","observation_id":"ae313d5e-4036-4968-8938-9941af6237c3","resolution":{"observed_at":"2026-08-11T00:42:54.719926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.703549Z","title":"Lightglue: Local feature matching at light speed","venue":null,"work_id":"65a1cca4-a9fa-4c26-b177-b35f1c3c459c","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.964039Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:3f11999511b27655207e77e890f02e8632d574cf2b545f0f16a3a01acc501043","observation_id":"19770a15-9bf3-48fb-a67f-2ac7a9e81226","resolution":{"observed_at":"2026-08-11T00:42:54.708013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.690938Z","title":"Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection","venue":null,"work_id":"cf5bb002-c67b-49dc-af1a-ec7cc21c9d89","year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.967689Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:432f63a8a72cf916b0b48218c726de17ad923bcfbad12dfe645ae11e96aa30a6","observation_id":"abc0940d-3c9b-4cd2-8900-58beb0c19925","resolution":{"observed_at":"2026-08-11T00:42:54.695253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.679378Z","title":"Paint trans- former: Feed forward neural painting with stroke prediction","venue":null,"work_id":"8c326924-21b3-405d-92ef-18d24aa7c75a","year":2021},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.971045Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:113ae4ba16c796f86dcf09c0e6338bf13c102841ba98523f0a8d5fcff24b4441","observation_id":"ecd93537-75f7-483c-848e-426bf8020ec6","resolution":{"observed_at":"2026-08-11T00:42:54.683518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.668181Z","title":"Grid: Guided refinement for detector-free multimodal image matching","venue":null,"work_id":"a643c961-a595-40f2-8350-17f860bec5a6","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.974525Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:7118f7190095f8970fa1cc0323c3d1ef6873f4c6a8033dafab1dcf6c9faab311","observation_id":"468fb831-c1eb-484a-aedc-be0173259d6a","resolution":{"observed_at":"2026-08-11T00:42:54.672022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.657488Z","title":"Image matching from handcrafted to deep fea- tures: A survey","venue":null,"work_id":"74e66707-256f-4167-b3f8-96d15fb0a372","year":2021},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.977842Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:413725b6697ffd1c2b3e2cd4750265f7cce96807461c91d0c3340edccefedbe9","observation_id":"046edb7c-bd7d-4dbb-b402-515e4f7c9fb3","resolution":{"observed_at":"2026-08-11T00:42:54.661193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.646449Z","title":"Dinov2: Learning robust visual features without supervision","venue":null,"work_id":"e83d8c58-e365-49ea-9221-e6da7e458565","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.981232Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:8a95a78af98e7fe18b5bd96b7f7406d2f32959a416f8e95c28896e425d860f6d","observation_id":"4127f4f7-eecb-4833-a1dc-1f146cb181de","resolution":{"observed_at":"2026-08-11T00:42:54.650148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:53.984782Z","title":"From coarse to fine: Robust hierarchical localization at large scale","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.984782Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:c7a2f276ea3ee3b04f2635315fd4b3707a9a1755297e896d42c5875ddd6b4e39","observation_id":"77d9a065-7872-4b1b-ad2a-c49caebb1695","resolution":{"observed_at":"2026-08-11T00:42:53.984782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.628876Z","title":"Superglue: Learning feature matching with graph neural networks","venue":null,"work_id":"f26afc4f-a4e9-465f-8608-9ce2bc669d75","year":2020},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.988430Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:a10ad0687f854d6b0f4771d4d71c3979c17ad2514fb93bceab35d268bc7d3f46","observation_id":"e7e23413-a51b-4091-bd04-a16df6722672","resolution":{"observed_at":"2026-08-11T00:42:54.632515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.618156Z","title":"Benchmarking 6dof outdoor visual localization in changing conditions","venue":null,"work_id":"d5a9680d-4654-4bd8-865b-21ead6b79a06","year":2018},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.991713Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:a0864c7c141b007facd5d7973b10e45f7df964b52239fcd787a881729c4830ca","observation_id":"1c890bb6-3fcf-4388-8058-c09444497c4c","resolution":{"observed_at":"2026-08-11T00:42:54.621972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.607766Z","title":"Structure-from-motion revisited","venue":null,"work_id":"d8d8c1e5-a016-496b-830c-1b761800b662","year":2016},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.995266Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:b103b6dddaa42e19e4efc081b58f06607c80ef6ca7c7c0760daab9dddfed8aeb","observation_id":"9337629e-cc97-4e1c-9897-60de1df90536","resolution":{"observed_at":"2026-08-11T00:42:54.611680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.597213Z","title":"Pixelwise view selection for un- structured multi-view stereo","venue":null,"work_id":"02dd60d3-cfd1-43e9-8898-15293ffecda3","year":2016},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.998655Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:8fb338b149dc17a62d9582ac156801d50fe7e24af26c71ce8cc6ecd3911cbe1d","observation_id":"0bcea292-991e-4275-afd0-341df7d8e5a2","resolution":{"observed_at":"2026-08-11T00:42:54.601126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.002126Z","title":"pytorch-fid: FID Score for PyTorch","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.002126Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:8919d91d122d019c82c2bf98858b79cc066f246c05381f45e2d7926379ed0875","observation_id":"39036be2-e1a9-4d28-a1e6-a9598d61948d","resolution":{"observed_at":"2026-08-11T00:42:54.002126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.580635Z","title":"Gim: Learning generalizable image matcher from internet videos","venue":null,"work_id":"5194828e-78e2-476b-804c-8e0924144239","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.005768Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:0fe105b391df80eb36eabee3f8325a35e28d39853959bafaeb12f929d68155bd","observation_id":"6c69d802-bd87-4ee1-a3a5-8117554f9dd2","resolution":{"observed_at":"2026-08-11T00:42:54.584349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.569360Z","title":"Indoor segmentation and support inference from rgbd images","venue":null,"work_id":"9e388274-be4e-4e2c-b45b-26ac3def91f9","year":2012},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.008897Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:04ef754653139b086f3b72f623d64b8478609783731655bcbb5a87e0a4e7cb84","observation_id":"8efe92ce-ae44-4591-b24c-1df4afa03300","resolution":{"observed_at":"2026-08-11T00:42:54.574075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.557623Z","title":"Loftr: Detector-free local feature matching with transformers","venue":null,"work_id":"4bed8f96-92e4-4228-a225-4cfea19fe8b8","year":2021},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.161381Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:f3770fda14b44d52855056279545d6a6af40af6fff5f4c34baaa6dc4a386ed9b","observation_id":"f72717ea-8d2e-4bb7-bd91-9eab87a600bb","resolution":{"observed_at":"2026-08-11T00:42:54.562019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.546284Z","title":"Piafusion: A progressive infrared and visible im- age fusion network based on illumination aware.Information Fusion, 2022","venue":null,"work_id":"47841c6c-fa1b-4bb1-aec5-522870ac26fc","year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.165439Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:7693a0d6f2fd17482f4134a7b05656c36a3e754299dabfb5c3f3bc94b2add810","observation_id":"433c5a25-e3ec-4563-8b68-3d048d3ab478","resolution":{"observed_at":"2026-08-11T00:42:54.550119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.534814Z","title":"Xoftr: Cross-modal feature matching transformer","venue":null,"work_id":"85576dff-6996-41af-8566-67ecc63d8033","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.168776Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:395cdc1f53a939a92011a82e3599edd1de9e9b7da7a6a6e7a7ece1dab0b03779","observation_id":"48d3f03f-c78d-47a4-9e76-8ff69d56a7d3","resolution":{"observed_at":"2026-08-11T00:42:54.538684Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.00463","last_updated":"2019-08-29T04:17:49Z","snapshot_observed_at":"2026-08-15T10:05:41.585216Z","submitted_at":"2019-08-01T15:39:54Z","title":"DIODE: A Dense Indoor and Outdoor DEpth Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.00463","snapshot_observed_at":"2026-08-11T00:42:54.172553Z","title":"Diode: A dense indoor and outdoor depth dataset","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.172553Z"},"links":{"cited_paper":"/paper/1908.00463","citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:73869b0e38aa6569828b03f779db4cf4ab85221ddda7f8f2907c3848e20e38ce","observation_id":"fb3dacfe-810d-4f90-9a89-ad83d39324ec","resolution":{"observed_at":"2026-08-11T00:42:54.172553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.523033Z","title":"Unsuper- vised misaligned infrared and visible image fusion via cross- modality image generation and registration","venue":null,"work_id":"9c143e1a-4358-4c71-9fed-a958b1536dfe","year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.176366Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:f5db74e22f4ce692d84186f0cfb04a07acb85fa970810b5d5bd3222230e6c983","observation_id":"14b42ee4-3719-4870-918c-8587d0185282","resolution":{"observed_at":"2026-08-11T00:42:54.527618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.512470Z","title":"Visev- ent: Reliable object tracking via collaboration of frame and event flows","venue":null,"work_id":"4f0bc824-2d19-4402-a909-d78d1fa694d6","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.179639Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:873a7de6a67e698195f3d824fede219e121c02cf872903a14750f4a59753dcbb","observation_id":"e3145c57-b25a-42a0-98a3-570c45bd2e8f","resolution":{"observed_at":"2026-08-11T00:42:54.516276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.501838Z","title":"Efficient loftr: Semi-dense local feature matching with sparse-like speed","venue":null,"work_id":"a967091e-ac4d-4c97-bae0-6d554a7c01ab","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.183293Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:9918d53dd9108057b14f5258afcb6dfc86878b83214c84f5217eeb6be0230468","observation_id":"6b62bf13-cf04-445f-b3f0-ef1b10737dda","resolution":{"observed_at":"2026-08-11T00:42:54.505596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.490914Z","title":"Image quality assessment: from error visibility to structural similarity","venue":null,"work_id":"0e097c26-f6c7-4da5-a6e7-4444bdb105a2","year":2004},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.186852Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:5058ea36a0986abc8361840e6e0d22dbb95c4fc3107a3d824f66b487eabe66b4","observation_id":"8fb35b4d-226c-444d-93e9-b73ac2698fde","resolution":{"observed_at":"2026-08-11T00:42:54.494736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.480328Z","title":"Single-model and any-modality for video ob- ject tracking","venue":null,"work_id":"efad7348-fe19-4798-a3a2-09cde331b046","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.190563Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:2b9861a723d2537eda0159e80341f79e194d60d7b2437100eaf18e199a702719","observation_id":"16eea2ac-dfc3-4470-a62c-8e77d19ba04e","resolution":{"observed_at":"2026-08-11T00:42:54.484151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.470197Z","title":"Adversarial open domain adap- tation for sketch-to-photo synthesis","venue":null,"work_id":"effc842e-c2a5-4121-a0c8-3799c7ec7fc9","year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.194084Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:7b708bcf17d222563ff934e9428a382108bc1393540c6b3e2ae798b2eadb6067","observation_id":"69c16280-a59b-4a1a-867c-e7c54abb08b9","resolution":{"observed_at":"2026-08-11T00:42:54.473976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.458140Z","title":"Murf: Mutually reinforc- ing multi-modal image registration and fusion","venue":null,"work_id":"506d8a85-ea05-4309-b989-9c19ab6b9df6","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.197559Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:06b3a8ed4d3fed27376c59ddc35ae0ec86e55c07fb3ef75b7b22423c53c748dc","observation_id":"0d60e52a-2d9e-40d0-a74a-92273c94241e","resolution":{"observed_at":"2026-08-11T00:42:54.462136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.447100Z","title":"Towards grand unified representation learning for unsupervised visible-infrared per- son re-identification","venue":null,"work_id":"1d55e917-3c61-4152-82c3-231a685d4ca2","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.200909Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:63319096aba812e034f44cfce61e8df74761d8e12b140ae6e5b515fab9b30f3e","observation_id":"f60e33dd-562d-4922-ab7e-895d40fd0538","resolution":{"observed_at":"2026-08-11T00:42:54.450956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.435598Z","title":"Depth anything: Unleashing the power of large-scale unlabeled data","venue":null,"work_id":"fc118909-0c8d-4f77-9f0f-c477eeef2527","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.204356Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:074df0c590538a605f269f0adf6ce4eb3e9ed51a1226387d5073460f13ed684c","observation_id":"70e8b3d2-702a-487d-ad76-d1bd27585028","resolution":{"observed_at":"2026-08-11T00:42:54.439912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.424498Z","title":"Depth any- thing v2","venue":null,"work_id":"058c30c9-65c9-4655-ba5d-30cc52222814","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.207811Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:795782b0ecb61502951aedaf44bb96f42c2fe25c367d77e5fc6f03712b3f4675","observation_id":"6b95e3d0-154a-4331-9a88-acb9738b0289","resolution":{"observed_at":"2026-08-11T00:42:54.428400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.413638Z","title":"Multi-modal remote sensing image matching considering co-occurrence filter","venue":null,"work_id":"000a79ab-3059-428c-bd6b-177a3c6ea807","year":2022},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.211144Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:eea1824f925e7a90f55a0b52f64c2896b5b5e7ea60d07fb51f1d6d698007737e","observation_id":"08563607-dbb7-47f3-b7c6-b97bc878727a","resolution":{"observed_at":"2026-08-11T00:42:54.417373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.402133Z","title":"Fast and robust matching for multimodal re- mote sensing image registration","venue":null,"work_id":"9efb003e-b764-4538-b7e5-33c128c8816f","year":2019},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.214519Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:8f7ff72bd9648736c95c97001bcf9965a384565410ddc7a7b2afd4c9bf002522","observation_id":"1eefbe6d-0084-461d-bd38-10092d7af402","resolution":{"observed_at":"2026-08-11T00:42:54.405978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.218126Z","title":"Image fusion meets deep learning: A survey and perspective","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.218126Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:e939a3b079ecbfddfa077e29f8d32ec30fed25265f5a5e6b567ad557ddc10052","observation_id":"e97b1a08-d091-4c1c-9500-43aae373833e","resolution":{"observed_at":"2026-08-11T00:42:54.218126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.384154Z","title":"Sparse-to-dense multimodal image registration via multi-task learning","venue":null,"work_id":"9550bde4-f669-497d-ae3e-9141cc562da3","year":2024},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.221331Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:55c00ae73bd06b0792d7e16079b09985888b920fcd637fda3afe99831608282a","observation_id":"b51c8b40-e605-46cc-b1cd-58aa1afc34dd","resolution":{"observed_at":"2026-08-11T00:42:54.388227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.224672Z","title":"The unreasonable effectiveness of deep features as a perceptual metric","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.224672Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:959981469f2b2e24a4322aac89728bc2f5cb633d0caba412690805c372c2cf58","observation_id":"b21e1cd7-7e48-4cc3-ba33-9d545303f62a","resolution":{"observed_at":"2026-08-11T00:42:54.224672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.366654Z","title":"Convmatch: Rethinking net- work design for two-view correspondence learning","venue":null,"work_id":"74b26cd0-ab92-41dd-acea-bbae6f1d6271","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.227876Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:1a127ceba6a7b63b98871ce29af3331baaa3b57193630f1ebae38a950eb8b809","observation_id":"a59ad04a-bbba-4f4a-b462-664bfe6bf400","resolution":{"observed_at":"2026-08-11T00:42:54.370457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.355867Z","title":"Diverse embedding ex- pansion network and low-light cross-modality benchmark for visible-infrared person re-identification","venue":null,"work_id":"21da36ee-aa6c-4573-af6f-29a829f80e02","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.231073Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:0d24336b29546ab9b9791cf85de9e3796150b2f8dfbf3082fb9dffd364340a93","observation_id":"893a8e54-3e95-4835-9985-2e90a7b83862","resolution":{"observed_at":"2026-08-11T00:42:54.359805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.344677Z","title":"Self- supervised pretraining via multimodality images with trans- former for change detection","venue":null,"work_id":"5526eefc-ec09-49ab-a1bb-92c26bc173a1","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.234308Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:847e29fee89c385b21661ef50aa3f2e573850c0b967b4c1c53cb99f11fdcd167","observation_id":"6f8f4224-745f-4074-bba0-8b8be0f7e9db","resolution":{"observed_at":"2026-08-11T00:42:54.348500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.333993Z","title":"Vm- loc: Variational fusion for learning-based multimodal cam- era localization","venue":null,"work_id":"28c863fc-7bf6-4ed2-8765-2125cd915904","year":2021},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.237655Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:ed42ca7c255bf510fcc1ff866e7fb9803dcd4cf766e6781e9552310da8774677","observation_id":"eabce429-cf05-4b69-beef-85f030fcf900","resolution":{"observed_at":"2026-08-11T00:42:54.337773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.323128Z","title":"Visual prompt multi-modal tracking","venue":null,"work_id":"a3251344-2319-4e96-9a93-5d90b3855250","year":2023},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.240964Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:ad12dba65d7c45504b02634db178fd98ddd70aefe8e57b3f2019c5202c80942f","observation_id":"ccc8a946-026b-4c03-acb8-723d063d7076","resolution":{"observed_at":"2026-08-11T00:42:54.326871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.311576Z","title":"XoFTR used a handcrafted method to transfer RGB to IR, while CPSTN is a cycle-consistent perceptual network","venue":null,"work_id":"95f85a46-c93d-4e6d-b6d6-2aa1ac067a91","year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.244556Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:f5bd8f52a06c06542e9bfae5d5ed1a4b3bf7ec559f00f22051ae2c01f27eb401","observation_id":"f0242c27-655b-4355-a1ce-aac9bddeaa93","resolution":{"observed_at":"2026-08-11T00:42:54.316219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.298769Z","title":"Bold indicates the best","venue":null,"work_id":"2b9cf110-bb36-4caf-bc70-a7c252ffea2a","year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:54.247849Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:1d537948a119d089674d828676c6ab12fe8562d4bd62d47b59550f39916ec72a","observation_id":"41291470-fd43-469e-bf62-5f4bac6bd38f","resolution":{"observed_at":"2026-08-11T00:42:54.303654Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T00:42:54.782063Z","title":null,"venue":null,"work_id":"fb33414b-eb96-4173-b334-1c9456942a24","year":null},"citing_paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T00:42:53.939209Z"},"links":{"citing_paper":"/paper/2412.19412"},"observation_digest":"sha256:30b62c5fe9d96d8ade894aff5bbdba982d23cf4eb3ab2ac088adc20e8c38e847","observation_id":"51b3fd85-b7e4-4cfa-9b67-ba1140aac15a","resolution":{"observed_at":"2026-08-11T00:42:54.785980Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.19412","last_updated":"2025-03-29T09:04:13Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-15T11:35:26.174813Z","submitted_at":"2024-12-27T02:39:50Z","title":"MINIMA: Modality Invariant Image Matching"},"reference_resolution":{"displayed":69,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":0,"verified_fuzzy":55},"total_outbound_references":69},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 69 of 69 outbound references and 0 inbound Pith citation observations for arXiv:2412.19412."}