{"as_of":"2026-08-08T12:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1ba3e7375885310db0c70fbae375b116d08c0905f4db43d1f8ca97d04d870b20","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:00:54.910555Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.03709/citation-record","integrity":"/paper/2506.03709/integrity","json":"/paper/2506.03709/citation-record.json","paper":"/paper/2506.03709"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.539630Z","title":"Syn2real domain generalization for underwater mine-like object de- tection using side-scan sonar","venue":null,"work_id":"5d588c0f-d0ba-4cb7-be5d-b09668f27788","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.740031Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:1a9c54ecdc27d38b28b7b7ce828a90f6018dffd921190e4c87f8abe689f01dda","observation_id":"61bcb3ed-c11e-41de-9292-9953c27c056d","resolution":{"observed_at":"2026-08-07T11:00:55.544548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.518704Z","title":"What a mess: Multi-domain evaluation of zero-shot semantic segmentation","venue":null,"work_id":"4278b932-db3b-42e4-9d20-c8d6b0e488c6","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.745956Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e42de78ad436f723f5c2eb5bc3444d6d9ad90c4ccb5dddfa58267503eaf95f08","observation_id":"76eaad89-6c37-4ea9-b870-15590b3a0629","resolution":{"observed_at":"2026-08-07T11:00:55.527529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.500418Z","title":"Coco- stuff: Thing and stuff classes in context","venue":null,"work_id":"815e8e44-1410-46e3-ab02-aef249596f26","year":2018},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.750519Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:136aa2cfcf9b6fa3e228ea2673c6b76cfcefe471fe8431e907f3f04b62f72ade","observation_id":"f1354a32-ad22-4205-8099-8c7d5cd78189","resolution":{"observed_at":"2026-08-07T11:00:55.505663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.755631Z","title":"Cat- seg: Cost aggregation for open-vocabulary semantic seg- mentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.755631Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e534e70f67dbd777e99271c4fdbc6735ea45157cb36b1c6eb19b955d671923b7","observation_id":"c040cc88-6947-4f53-9256-b894433ba3e1","resolution":{"observed_at":"2026-08-07T11:00:54.755631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.470595Z","title":"De- coupling zero-shot semantic segmentation","venue":null,"work_id":"f5093db3-55b2-473a-abaa-913812dfb95f","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.760788Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:f2183d973495f0b35d38c3e29b8539b8e4d8c7d3381aaf99b796aeab6da4fd52","observation_id":"8225a432-8fb9-4c19-862f-d9d99a7f709f","resolution":{"observed_at":"2026-08-07T11:00:55.475641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.765166Z","title":"Open- vocabulary panoptic segmentation maskclip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.765166Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:803924f617c5d6e88ea6311f4276d4e149ff34611ff22b606f849e45a6c4caf6","observation_id":"e34b6372-d23f-449d-9ade-7634d1f958e0","resolution":{"observed_at":"2026-08-07T11:00:54.765166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.442077Z","title":"The pascal visual object classes challenge: A retrospective","venue":null,"work_id":"186a5517-e262-461a-ba56-4739f58f6170","year":2015},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.770060Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:18207fa48e4950f50a9ac05581f8f05e4a5a227f1f3dff35744daf763c0de800","observation_id":"d2defca1-eca3-4294-8507-7965987b4cf6","resolution":{"observed_at":"2026-08-07T11:00:55.447014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.425096Z","title":"Scal- ing open-vocabulary image segmentation with image-level labels","venue":null,"work_id":"de8acfef-18be-4ccf-b2d9-12bdedf4caea","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.774420Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:1c6f2a71c7388ec42c3c56ff5307108ce0aba61d5f80549da89bd4fe64784561","observation_id":"9b8589ed-ec1e-42b0-978d-9ee7e4c0c37f","resolution":{"observed_at":"2026-08-07T11:00:55.430009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.402405Z","title":"Multispectral video se- mantic segmentation: A benchmark dataset and baseline","venue":null,"work_id":"cbcee5d1-048b-4305-9f28-37f3956824a6","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.778848Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:1e33d1c34f22287291c0c8973e2c87fb641a07225546966745d0786bb0f19162","observation_id":"e6b3209c-6663-4ac3-a950-a2d85c81a5be","resolution":{"observed_at":"2026-08-07T11:00:55.413145Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.783608Z","title":"Scaling up visual and vision-language representa- tion learning with noisy text supervision","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.783608Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:a1fba82bde4efec02e9f6b92af056838eae85604475aac4a761b24782faf26ca","observation_id":"06b53928-1b0a-4d3b-bc2f-cd6a3eee769d","resolution":{"observed_at":"2026-08-07T11:00:54.783608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.373642Z","title":"An efficient approach with dynamic multiswarm of uavs for forest firefighting","venue":null,"work_id":"04d0f5e3-a0f7-408a-9d35-73f7509b2931","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.788414Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:aeeacd2c339fe20d035bc562f73003ef7bb8eb8af14f5aa6fb012c5b7c66a670","observation_id":"d7709b8c-c5ef-4dcd-a154-628bb19fb673","resolution":{"observed_at":"2026-08-07T11:00:55.378331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.356712Z","title":"A resource-efficient decentralized sequential planner for spa- tiotemporal wildfire mitigation","venue":null,"work_id":"1222749b-0168-468e-8495-ff7779f48cb3","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.792908Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:93b6159a323e44c80105c2c0abf20148fa21b308de3e38eeffbcc1b8cb8c7a0b","observation_id":"b71fb33b-2ef7-4282-a8eb-dc0dad063e5d","resolution":{"observed_at":"2026-08-07T11:00:55.362057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.10054","last_updated":"2022-02-21T09:03:34Z","snapshot_observed_at":"2026-08-04T02:30:25.691953Z","submitted_at":"2022-02-21T09:03:34Z","title":"Fine-Tuning can Distort Pretrained Features and Underperform Out-of-Distribution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.10054","snapshot_observed_at":"2026-08-07T11:00:54.797427Z","title":"Fine-tuning can distort pretrained fea- tures and underperform out-of-distribution","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.797427Z"},"links":{"cited_paper":"/paper/2202.10054","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:426f347328d088541d1c74297231a2a4a07931f0403bc7382b47217937d195c8","observation_id":"3b7a43fd-23ec-47e4-bee7-cee29268ed97","resolution":{"observed_at":"2026-08-07T11:00:54.797427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.338122Z","title":"Caltech aerial rgb-thermal dataset in the wild","venue":null,"work_id":"3b696df5-da0f-48d5-a084-214bb73ec595","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.802076Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e2f789d89c1bee0e9a0f630833d7f591c8792de3893a6be66424ca3ba044c7d8","observation_id":"d925f3fa-41e5-42e1-b517-93bc93e31d41","resolution":{"observed_at":"2026-08-07T11:00:55.342852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.321652Z","title":"Open-vocabulary semantic segmentation with mask-adapted clip","venue":null,"work_id":"f9137dbd-f246-453f-bfd8-35dd397d0a8c","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.806584Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:d5ee15b3cfa1f40da2d818699f0f28bf54b58d7943446a5c90a40a00153e74fb","observation_id":"26dfce84-5c02-4655-82dd-55a0c1e62a1e","resolution":{"observed_at":"2026-08-07T11:00:55.326883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-07T11:00:54.811123Z","title":"Holistic evalu- ation of language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.811123Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e87455aed8065facaf404adca7d48e1fa74607dea262438e8ba243ba42a7f2ff","observation_id":"734ea006-f8da-425d-8a43-001721b94a15","resolution":{"observed_at":"2026-08-07T11:00:54.811123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.306240Z","title":"Uavid: A semantic segmentation dataset for uav imagery","venue":null,"work_id":"e43a1b0c-200d-457c-bc10-dc582e87d44f","year":2020},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.816141Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:979a5f750b3fae18a381ba63172acb32131c37e9974a3bef511784aaa1dd7b77","observation_id":"6df921a1-10ad-4f78-a4fc-df9e1ce36f49","resolution":{"observed_at":"2026-08-07T11:00:55.310869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.820409Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.820409Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:a695e914da6a95e2ab53a98d8aa2f65b58d5f6d04eb10304d43cb29c21c72c8c","observation_id":"1986572b-a1de-4181-ad50-be376089ffeb","resolution":{"observed_at":"2026-08-07T11:00:54.820409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16067","last_updated":"2024-07-22T21:54:19Z","snapshot_observed_at":"2026-07-06T18:50:22.540341Z","submitted_at":"2024-07-22T21:54:19Z","title":"LCA-on-the-Line: Benchmarking Out-of-Distribution Generalization with Class Taxonomies","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16067","snapshot_observed_at":"2026-08-07T11:00:54.824912Z","title":"Lca-on-the-line: Benchmarking out-of-distribution generalization with class taxonomies","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.824912Z"},"links":{"cited_paper":"/paper/2407.16067","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:b64800f8b3fc8126bb6ceb36571000041492e09ac2621dcc89e0f347675255ad","observation_id":"067a8bb2-80b0-42a1-9339-c6959c4d70c3","resolution":{"observed_at":"2026-08-07T11:00:54.824912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.829772Z","title":"Fully complex-valued fully con- volutional multi-feature fusion network (fc 2 mfn) for build- ing segmentation of insar images","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.829772Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:7e07601dd95d9c3e38b72967cc1c922689178181fee707b6c44f978bb812e74c","observation_id":"62af2453-9620-4cfe-a25e-31f460f59953","resolution":{"observed_at":"2026-08-07T11:00:54.829772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.261381Z","title":"Deepmao: Deep multi-scale aware over- complete network for building segmentation in satellite im- agery","venue":null,"work_id":"dbc2ade5-7208-4dc1-9d58-50a128de11b7","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.835668Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:ce9aa02684d599ce33d82014e762311eab0688b04f5e2c01d01c64d1d6fc2667","observation_id":"4ccb78e6-1308-460d-8d51-125ba0a0f238","resolution":{"observed_at":"2026-08-07T11:00:55.267194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.243540Z","title":"Ssl-rgb2ir: Semi-supervised rgb-to-ir image-to-image translation for enhancing visual task train- ing in semantic segmentation and object detection","venue":null,"work_id":"34a89bb0-a484-418e-b5db-0c6297563088","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.840306Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e6e14ad56c78c46389a085c61bc2e5c00af7ed815defd310d87d073f38c4c66e","observation_id":"6e874028-ec58-4e5d-add3-6f0fcd8580c3","resolution":{"observed_at":"2026-08-07T11:00:55.248554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.227038Z","title":"Skd- net: Spectral-based knowledge distillation in low-light ther- mal imagery for robotic perception","venue":null,"work_id":"1bad1597-ebd1-4e6e-b01a-d8f6cd22ef1e","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.844861Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:51a7434b38b499e8d1b759996f2bc9d8282035c45aa73f8ff31bb629ee3cb695","observation_id":"769d20ca-7863-4742-95e1-78745631a294","resolution":{"observed_at":"2026-08-07T11:00:55.232254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15728","last_updated":"2025-04-22T09:22:11Z","snapshot_observed_at":"2026-08-07T16:00:18.165129Z","submitted_at":"2025-04-22T09:22:11Z","title":"SAGA: Semantic-Aware Gray color Augmentation for Visible-to-Thermal Domain Adaptation across Multi-View Drone and Ground-Based Vision Systems","version":1},"cited_work":{"arxiv_id":"2504.15728","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.15728","snapshot_observed_at":"2026-08-07T11:00:54.969171Z","title":"SAGA: Semantic-Aware Gray color Augmentation for Visible-to-Thermal Domain Adaptation across Multi-View Drone and Ground-Based Vision Systems","venue":"cs.CV","work_id":"5f1f6e26-c77d-4f40-8ac2-0c57e033721c","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.849654Z"},"links":{"cited_paper":"/paper/2504.15728","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:e801b75bc72012a96f9d316f9ffeef5018d5d8548946c16801221ac0d1aef664","observation_id":"8cc2452f-7b70-46a4-b991-7dde6f189603","resolution":{"observed_at":"2026-08-07T11:00:54.976223Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.209872Z","title":"Ogp- net: Optical guidance meets pixel-level contrastive distilla- tion for robust multi-modal and missing modality segmen- tation","venue":null,"work_id":"fb8ef95e-b6ac-4a1e-8f68-acb172fa8013","year":2025},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.854430Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:2000188133aff6450c18065639ec6402a2fb949bb45d72364512c9250c5ae82b","observation_id":"8ce25091-7025-4975-9e0b-153017e6e999","resolution":{"observed_at":"2026-08-07T11:00:55.215098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.191934Z","title":"Isprs potsdam dataset within the isprs test project on urban classification, 3d building reconstruction and semantic labeling, 2012","venue":null,"work_id":"a481a4b7-1c30-4925-9d3b-203b27fa6f08","year":2012},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.858663Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:160c1470df21afbaf5a7f60d27dd4a991c104df5987751b3fe9147f689f54110","observation_id":"ee30a712-b673-4314-8d98-5b34950791ca","resolution":{"observed_at":"2026-08-07T11:00:55.197910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.175587Z","title":"Piafusion: A progressive infrared and visible im- age fusion network based on illumination aware.Information Fusion, 83:79–92, 2022","venue":null,"work_id":"aaa7ff63-887f-4af2-a6cc-94a0305ae88a","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.862947Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:cb0d3f228084839ccd873e91837c14a72f64201859d58d836e046e5f782d6970","observation_id":"cb27bcad-15a0-4ec2-960f-e1b7af728c17","resolution":{"observed_at":"2026-08-07T11:00:55.180836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.867104Z","title":"Mrfp: Learning generalizable semantic segmentation from sim-2-real with multi-resolution feature perturbation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.867104Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:906995a11d94abe946e4b2bf3d0b47e0994a368284097b13b80561152383cd0d","observation_id":"421215af-8ad4-40dc-b6ac-f8264a09775b","resolution":{"observed_at":"2026-08-07T11:00:54.867104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.871492Z","title":"Robust fine-tuning of zero-shot models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.871492Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:9b98923745f9ba1c44f834ac4c23a21ed2d90a9294b092321debc75c65583012","observation_id":"b2fb9c41-8b64-4f50-8232-ff423100053e","resolution":{"observed_at":"2026-08-07T11:00:54.871492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.875749Z","title":"Open-vocabulary panop- tic segmentation with text-to-image diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.875749Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:b5a747db06f191a2425960d9cdd98c5ac239e088fe0ee021705e9530207c8145","observation_id":"733ca119-7afd-4c32-ab4b-5ca1414260bd","resolution":{"observed_at":"2026-08-07T11:00:54.875749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.127274Z","title":"A simple baseline for open- vocabulary semantic segmentation with pre-trained vision- language model","venue":null,"work_id":"cfaf407d-0dcd-4b1a-a952-9de27d2f2586","year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.880064Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:88c4287b24ae231a6f9248290fc9fcd355d8e1ed8d37aec43e32bbda6ae85725","observation_id":"d3158162-599b-40c0-bdc4-3f476c4dc7bd","resolution":{"observed_at":"2026-08-07T11:00:55.131905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.111447Z","title":"Side adapter network for open-vocabulary semantic segmentation","venue":null,"work_id":"2242bb49-0541-49ba-a9d4-0c38279b9d34","year":2023},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.884328Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:74125cb0ae3216d7f17b151bf4213f4c568d61d918b4e64463ead1b84ffa88e8","observation_id":"2e0849b6-a601-4887-a56d-6f4e2d2eff9e","resolution":{"observed_at":"2026-08-07T11:00:55.116220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.888526Z","title":"Convolutions die hard: Open-vocabulary seg- mentation with single frozen convolutional clip","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.888526Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:0e94b34972fe2728d093cf0aefe84315b1f65457268b08af56a96dda2134b327","observation_id":"a81cd960-f265-4d4f-b8ae-2be8a902effd","resolution":{"observed_at":"2026-08-07T11:00:54.888526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11432","last_updated":"2021-11-22T18:59:55Z","snapshot_observed_at":"2026-07-06T12:11:02.119174Z","submitted_at":"2021-11-22T18:59:55Z","title":"Florence: A New Foundation Model for Computer Vision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.11432","snapshot_observed_at":"2026-08-07T11:00:54.892839Z","title":"Florence: A new foundation model for computer vision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.892839Z"},"links":{"cited_paper":"/paper/2111.11432","citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:b60bee6f82c6786c8ff9ed13e142c0045346a1d965afc1a5b26cd5030a2a4f95","observation_id":"6ccf68ea-ae7e-4348-8e1b-26ffb316dae1","resolution":{"observed_at":"2026-08-07T11:00:54.892839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.080456Z","title":"Dept: Decoupled prompt tuning","venue":null,"work_id":"b0c4dfba-cb58-43ee-a6e6-b7e58d3ecc5b","year":2024},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.897416Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:a3dbce1aa0b2abd733d94834bdb392eee88593039dd5f19fdf80f57b4734ad7a","observation_id":"f4e0a183-1163-4ecf-8f73-f270a7e94a20","resolution":{"observed_at":"2026-08-07T11:00:55.085451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:55.064514Z","title":"Semantic under- standing of scenes through the ade20k dataset","venue":null,"work_id":"4236ba84-bbff-4276-81d0-88ab48cc60b4","year":2019},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.901790Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:0fe4cde3dfda75b94cd5da621b841caed9daeea92750c625c8984d542dd007a8","observation_id":"d905bb80-3118-487a-bd14-c44d85bafd18","resolution":{"observed_at":"2026-08-07T11:00:55.069426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.906244Z","title":"Extract free dense labels from clip","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.906244Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:2702aac1b16bff6bb376d65b3fbd90ccef7514ef969d69444b09c004f8450888","observation_id":"0738a1bc-72a2-489a-8a01-338c1c816f09","resolution":{"observed_at":"2026-08-07T11:00:54.906244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:00:54.910555Z","title":"Generalized decoding for pixel, image, and lan- guage","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.910555Z"},"links":{"citing_paper":"/paper/2506.03709"},"observation_digest":"sha256:0d0f136fda694ae9d277ef53b9ad4af8b9ca8f9355700686ff76f0e098d68b0f","observation_id":"c03845bb-6da7-42aa-9abe-26da37c6d0dc","resolution":{"observed_at":"2026-08-07T11:00:54.910555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.03709","last_updated":"2025-06-04T08:41:19Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T10:54:39.145107Z","submitted_at":"2025-06-04T08:41:19Z","title":"AetherVision-Bench: An Open-Vocabulary RGB-Infrared Benchmark for Multi-Angle Segmentation across Aerial and Ground Perspectives"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":1,"verified_fuzzy":22},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 0 inbound Pith citation observations for arXiv:2506.03709."}