{"as_of":"2026-08-09T06:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e675da8c4bacbf9c89323b3f05c2ca153f7d431ea26e7a94be6d1357c6ebeca0","coverage":[{"denominator":47,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":47,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T11:33:44.814543Z","state":"measured"},{"denominator":47,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":47,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.19857/citation-record","integrity":"/paper/2607.19857/integrity","json":"/paper/2607.19857/citation-record.json","paper":"/paper/2607.19857"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.010000Z","title":"The small-drone revolution is coming—scientists need to ensure it will be safe,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.010000Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:b924bd4fcdba467a6e4532565435a831a5327e664397be7c340142f57c335944","observation_id":"9b995d4e-2191-4b15-b83f-b8fc2466296d","resolution":{"observed_at":"2026-08-01T11:33:41.010000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.077507Z","title":"Champion-level drone racing using deep reinforcement learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.077507Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:b5dc002083c6550103c2c617dc946fe67f25b6fc8b27abb0c207218858afdf0b","observation_id":"49b07a8c-e707-4b7c-9a55-2c743b824f6b","resolution":{"observed_at":"2026-08-01T11:33:41.077507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.210959Z","title":"Video object segmentation without tem- poral information,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.210959Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:b4650adff20d00834b5c87e612760db01e599a72db1a6a50ae44c0fbc6697a16","observation_id":"761e62ab-1b1c-4192-b68f-2dad832a8038","resolution":{"observed_at":"2026-08-01T11:33:41.210959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.290989Z","title":"3d question answering for city scene understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.290989Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:33f349a4cbb90f655fea2a21e9e86a39918a1077320d4ebcb030b5f0503337ea","observation_id":"5c39917d-b99c-406e-b8f3-61d24b26056c","resolution":{"observed_at":"2026-08-01T11:33:41.290989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.391352Z","title":"Phenobench: A large dataset and benchmarks for semantic image interpretation in the agricultural do- main,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.391352Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:65bd27b28dd6408c29845bbb086d102420f30dbe763a73091ca7b7ccc218eb18","observation_id":"43efa3e9-286a-4e1b-be86-a3197c84754d","resolution":{"observed_at":"2026-08-01T11:33:41.391352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.479024Z","title":"Detecting flying objects using a single moving camera,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.479024Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:e57da6da69b3707a2faa739a5fee7c99dbfadaa2fcc051585b66d00f2c1b696a","observation_id":"01ecd4ed-e87b-4b06-8dec-7ad48ad22ef2","resolution":{"observed_at":"2026-08-01T11:33:41.479024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.574070Z","title":"Revisiting image-language networks for open-ended phrase detection,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.574070Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:1d0b9d0c874add21a23830e60a69042e3e20054f2c003cf0f688810083c98880","observation_id":"4f4d291c-5418-4571-8780-b59da9a6df49","resolution":{"observed_at":"2026-08-01T11:33:41.574070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.647764Z","title":"Mevis: A large- scale benchmark for video segmentation with motion expressions,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.647764Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:167b19a44831165c3b3a2b4e168ba1f93969717f7ea683718a012822d51715d8","observation_id":"d2b93844-3363-4b7a-aff3-0037efbdcd01","resolution":{"observed_at":"2026-08-01T11:33:41.647764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.750581Z","title":"Lamot: Language- guided multi-object tracking,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.750581Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:d772ff2cc0b1a135e039e563644964c447e8acec40126863166595e31463f05d","observation_id":"d5431214-f946-448e-974c-5886b8e16458","resolution":{"observed_at":"2026-08-01T11:33:41.750581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.793599Z","title":"Skyfind: A large-scale benchmark unveiling referring expres- sion comprehension for uav","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.793599Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:b6f42554aa6aea71390a80d9ce324a95cae838ffe539cccc49419507ef1fbaaa","observation_id":"a6822768-0774-409a-b3c4-642633a0a9f6","resolution":{"observed_at":"2026-08-01T11:33:41.793599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.860865Z","title":"Aerialmind: Towards referring multi-object tracking in uav scenarios,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.860865Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:a9fee240eb690764b306ba76d6382d9268a0099b33f68d6587bbd8820d690cea","observation_id":"50c4b8f7-f820-4c1e-96ac-d23aae4cef2d","resolution":{"observed_at":"2026-08-01T11:33:41.860865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:41.918717Z","title":"Event-aware instructed assistant for referring video segmentation,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:41.918717Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:416b733d751892ad6cfc2877a16d3fcc0f56923d4ef6fcfbce6258629ffa36b3","observation_id":"6da0cf3b-dd9b-4492-acd3-8afc41de44ce","resolution":{"observed_at":"2026-08-01T11:33:41.918717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.000557Z","title":"City-vlm: Towards multidomain perception scene understanding via multimodal incomplete learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.000557Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:6c033557de1d71f0020f791d6287e9545ce8eb1a5da7de5d1dbe6376d48cae19","observation_id":"cd07ae9b-fc66-4d12-97e5-44f84799e2fb","resolution":{"observed_at":"2026-08-01T11:33:42.000557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.085352Z","title":"Qwen2.5 technical report,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.085352Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:5155eebdd6069197648c1f738a7e9c842f30ee349e7eb1bfc63636ac973bd174","observation_id":"71600a21-e282-4709-8a33-42faab77c5fb","resolution":{"observed_at":"2026-08-01T11:33:42.085352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.16719","last_updated":"2026-03-28T16:54:56Z","snapshot_observed_at":"2026-08-07T05:59:39.049027Z","submitted_at":"2025-11-20T18:59:56Z","title":"SAM 3: Segment Anything with Concepts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.16719","snapshot_observed_at":"2026-08-01T11:33:42.259681Z","title":"Sam 3: Segment anything with concepts,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.259681Z"},"links":{"cited_paper":"/paper/2511.16719","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:b9c92ee7cc5d2ab8a89d61dc13216c9f576b844d394ddbe06048850ca06502bc","observation_id":"366e8549-1cbc-4afa-984b-b330d1f213b2","resolution":{"observed_at":"2026-08-01T11:33:42.259681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.350558Z","title":"React: Streaming video analytics on the edge with asynchronous cloud support,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.350558Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:2d31836bec4df403fc0c9c612678098e2aad4224b4bb432586d8907ead8ceada","observation_id":"305d387f-576c-47d0-99f8-484771a6ffed","resolution":{"observed_at":"2026-08-01T11:33:42.350558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.439765Z","title":"Multiple-object-tracking algorithm based on dense trajectory voting in aerial videos,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.439765Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:ab0fffac0cad5b772ce88f2d418db2046ffdcd081689e65c9b37d19f89fc8764","observation_id":"5b29eab7-6a4d-4271-8344-f237d491cea0","resolution":{"observed_at":"2026-08-01T11:33:42.439765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.515394Z","title":"Urvos: Unified referring video object segmentation network with a large-scale benchmark,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.515394Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:0d8e1f56e7cb7dd892f2b6de47f953c869cabf11a85cae55e2d0992b11e7816f","observation_id":"08fa439c-bdb6-4acf-8ed9-9f15f82ab321","resolution":{"observed_at":"2026-08-01T11:33:42.515394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.604426Z","title":"Sharegpt4video: Improving video understanding and generation with better captions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.604426Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:0362b6c5b5b748c76fdb03b42ab443adfafb248a950a3d085ebcdad40fa9b5de","observation_id":"b14fc791-78fa-4110-bcd4-66ae0c8a6cf8","resolution":{"observed_at":"2026-08-01T11:33:42.604426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.675678Z","title":"Mevis: A multi-modal dataset for referring motion expression video segmentation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.675678Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:c348d4e94f28edac6ec623f37de0d4f7f928d3d1a916a0479e471de7adaf5080","observation_id":"79b21db3-821c-4694-8687-61f39cbdae15","resolution":{"observed_at":"2026-08-01T11:33:42.675678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.778744Z","title":"Jtd-uav: Mllm-enhanced joint tracking and description framework for anti-uav systems,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.778744Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:efdb57c324f3d6e3bbdd5c5865bc68997034243198932701741bf02268cc0adb","observation_id":"844c35ca-a119-472a-8294-da0f114f5e6f","resolution":{"observed_at":"2026-08-01T11:33:42.778744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.829375Z","title":"Visa: Reasoning video object segmentation via large lan- guage models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.829375Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:d048246dd4094474677de70da36dd2c6ab3eea152fd258c11738a39137afa8b5","observation_id":"0e892caa-5ce7-4f56-bdb7-f50fa6c94cd4","resolution":{"observed_at":"2026-08-01T11:33:42.829375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-08T01:58:42.644918Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-08-01T11:33:42.889990Z","title":"Sa2va: Marrying sam2 with llava for dense grounded understanding of images and videos,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.889990Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:b08dd6a5a8993ebb7de419fe689e3dbd710c2af671b6de765a5f41a45a0816da","observation_id":"eed37ea5-009d-4755-b182-9cc0b34b6b93","resolution":{"observed_at":"2026-08-01T11:33:42.889990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:42.985730Z","title":"Unipixel: Unified object referring and segmentation for pixel-level visual reason- ing,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.985730Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:849666b80cb70f111f9088f75a61742e560ad947b40bfead46803acd06b1fec9","observation_id":"26c53c5e-2219-4323-a01b-152283af7035","resolution":{"observed_at":"2026-08-01T11:33:42.985730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-01T11:33:43.065979Z","title":"Sam 2: Segment anything in images and videos,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.065979Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:667d488e9f73f6ba7bcf167121cb6648850e1d22ca8d7dac3863176aa87c1ff5","observation_id":"166f5840-37de-4732-b725-fca92a964b90","resolution":{"observed_at":"2026-08-01T11:33:43.065979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.135856Z","title":"Videoglamm: A large multimodal model for pixel-level visual JOURNAL OF LATEX CLASS FILES, VOL. 14, NO. 8, AUGUST 2021 14 grounding in videos,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.135856Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:c90e1dd433f074b4f31e598dd483d99eaabc25d42ddbf2db6d8277c1d921d54c","observation_id":"4af657c0-610a-4099-a718-52299e96ebc4","resolution":{"observed_at":"2026-08-01T11:33:43.135856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.233507Z","title":"Glus: Global-local reasoning unified into a single large language model for video segmentation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.233507Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:6dd1f21d420d9a6ec8b2e9a5506fb003403617a592205fa176de50c801173103","observation_id":"7a0db379-3b2a-4df5-b8ca-796f9e2f9a06","resolution":{"observed_at":"2026-08-01T11:33:43.233507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.328278Z","title":"The devil is in temporal token: High quality video reasoning segmentation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.328278Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:9eb13c135564c77d53b21b8ebd999955662285bbc0c1ee2b961fdd2036218f1d","observation_id":"5c1cf94f-b519-42a1-b153-da195082896f","resolution":{"observed_at":"2026-08-01T11:33:43.328278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.435658Z","title":"Instructseg: Unifying instructed visual segmentation with multi-modal large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.435658Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:8ca86faeb62427c258b892143ff1adfefb28a9b865f510effae38126ef8049c5","observation_id":"1fc75169-9d8b-4957-8153-a284702ec0b5","resolution":{"observed_at":"2026-08-01T11:33:43.435658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.532666Z","title":"Geochat: Grounded large vision-language model for remote sensing,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.532666Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:0abd436e447e06ec80e9ff191846419b631cb032c1bd1bae3a8e263f8630f515","observation_id":"df21c549-3a59-43ec-94da-f40429c71d43","resolution":{"observed_at":"2026-08-01T11:33:43.532666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10100","last_updated":"2024-07-08T04:33:37Z","snapshot_observed_at":"2026-07-06T18:31:04.425917Z","submitted_at":"2024-06-14T14:57:07Z","title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10100","snapshot_observed_at":"2026-08-01T11:33:43.613930Z","title":"Skysensegpt: A fine-grained instruction tuning dataset and model for remote sensing vision-language understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.613930Z"},"links":{"cited_paper":"/paper/2406.10100","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:efe3b9bbd240a46d695d11ed6749704c554b04baccb16e4704c00cfb144951b2","observation_id":"aa2add77-efb9-4878-be89-fdcd1b629294","resolution":{"observed_at":"2026-08-01T11:33:43.613930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.690981Z","title":"Earthgpt: A universal multimodal large language model for multisensor image comprehension in remote sensing domain,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.690981Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:9119cbe2e1d7b796f1dab25ccf9a049f7016b307760f146c28f4c316c1dd27b9","observation_id":"2ef00eab-20ef-407f-b9a5-a1423abd28ee","resolution":{"observed_at":"2026-08-01T11:33:43.690981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-01T11:33:43.758425Z","title":"Kimi k2. 5: Visual agentic intelligence,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.758425Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:0844eb6cd36263c5ae364005d4a2186c5c5e535b64e9e71587b04a633ff5fde4","observation_id":"1b7777f8-6659-4ffb-9d4d-397c763d51cd","resolution":{"observed_at":"2026-08-01T11:33:43.758425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.831961Z","title":"Qwen3.6-Plus: Towards real world agents,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.831961Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:96baeb94e76cd82cb8847fec6b116d9ac6362ad8c4935c53a29da411ce17b753","observation_id":"beedfb6f-dfe7-4f2b-9f22-bf52ce0350a9","resolution":{"observed_at":"2026-08-01T11:33:43.831961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:43.921225Z","title":"Visual instruction tuning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.921225Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:eef518d0d0af277725df99e105d8df32b409f704152704861bd16e786bb595fa","observation_id":"211f0754-f661-4860-b3d8-3eb118d94675","resolution":{"observed_at":"2026-08-01T11:33:43.921225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-01T11:33:43.990599Z","title":"Qwen2.5-vl technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:43.990599Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:992be9f4ccdf661f03e0f77af90a56e230163369988e127d31d625009589c67d","observation_id":"49e040f5-7cb0-4289-9175-9dbffc0da9c2","resolution":{"observed_at":"2026-08-01T11:33:43.990599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.09608","last_updated":"2025-10-10T17:59:58Z","snapshot_observed_at":"2026-08-08T02:53:06.532732Z","submitted_at":"2025-10-10T17:59:58Z","title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.09608","snapshot_observed_at":"2026-08-01T11:33:44.071908Z","title":"Streamingvlm: Real-time understanding for infinite video streams,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.071908Z"},"links":{"cited_paper":"/paper/2510.09608","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:0211c5e387eae8ec27f2bbe2fcdd3ea9037c4c677150c22fa6824a939570461a","observation_id":"e7b2fe0a-3f59-41ca-9521-652d554e1f04","resolution":{"observed_at":"2026-08-01T11:33:44.071908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.133296Z","title":"A fast and accurate one-stage approach to visual grounding,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.133296Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:166c349aa0c95ae0ecbb9c3a94f3f10fbb9d075c2949e7996131d58d72678fbd","observation_id":"9f512ed1-8155-4f28-bd60-74bb222a0f54","resolution":{"observed_at":"2026-08-01T11:33:44.133296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.252921Z","title":"Improving one-stage visual grounding by recursive sub-query construction,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.252921Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:5e4328213e26116f098d6cad3ad3a4b0c25b3824946768cb7ba1994206aabddd","observation_id":"dfd9d33c-78e5-40bc-b1bf-c29bba6295cf","resolution":{"observed_at":"2026-08-01T11:33:44.252921Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.355563Z","title":"Referring transformer: A one-step approach to multi-task visual grounding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.355563Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:ef4bf3597172941401e33077c4a4916d2807a562456a6b03c2a4fe5e9538b90a","observation_id":"994a86ff-27c6-4364-b2da-a43282072bfd","resolution":{"observed_at":"2026-08-01T11:33:44.355563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.438359Z","title":"Transvg: End-to-end visual grounding with transformers,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.438359Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:81517673b0b3d07169cb7f16f4b46187845990d0f14a6a8dcb8653fee1fb810a","observation_id":"ed070ee0-2232-42c8-b209-4f4b6c51f151","resolution":{"observed_at":"2026-08-01T11:33:44.438359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.518896Z","title":"Improving visual grounding with visual-linguistic verification and iterative reasoning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.518896Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:a8247108b6eb59bb5ea7c87e13350ae459fe95b30044a8738896e1fb6d093143","observation_id":"a3617427-5543-4c34-81cf-13884fbc05a6","resolution":{"observed_at":"2026-08-01T11:33:44.518896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.607850Z","title":"Seqtr: A simple yet universal network for visual grounding,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.607850Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:f3fc4d0edd91f42257aa7d1b6c4b6c31f98184b72941d99d913eea0dc733e45f","observation_id":"15da3c5a-576d-4eba-b291-7e1adf5e5b66","resolution":{"observed_at":"2026-08-01T11:33:44.607850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.682676Z","title":"Shifting more attention to visual backbone: Query-modulated refinement networks for end-to-end visual grounding,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.682676Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:84adac8a3ac858cc157a9c4c52a12ccfa0c71e2f622e7597da3878b1a0dd4272","observation_id":"2c406c7c-c960-4212-b552-37e5b05d6790","resolution":{"observed_at":"2026-08-01T11:33:44.682676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.745993Z","title":"A survivor in the era of large- scale pretraining: An empirical study of one-stage referring expression comprehension,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.745993Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:10446388985dc2401fdec2519f07bca73e7168feb7027687fbb50190e6a1942e","observation_id":"8f6255a7-0317-45e2-98b1-3e35f30b41d7","resolution":{"observed_at":"2026-08-01T11:33:44.745993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T11:33:44.814543Z","title":"Polyformer: Referring image segmentation as sequential polygon generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:44.814543Z"},"links":{"citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:2b806d80ed09d62a2411006777a9b238f754ea2d3b7ccf44453a510fe9e044a6","observation_id":"2bc27e84-0bb9-4c6c-9f13-e9a704e31956","resolution":{"observed_at":"2026-08-01T11:33:44.814543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-01T11:33:42.188553Z","title":"Available: https://arxiv.org/abs/2412.15115","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:42.188553Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2607.19857"},"observation_digest":"sha256:9c8211e4304e518b9f54d55e244e464e0518f51da15b438f6e6eb4413780783c","observation_id":"c2c19032-b299-4504-a5f6-cad0810c8554","resolution":{"observed_at":"2026-08-01T11:33:42.188553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.19857","last_updated":"2026-07-22T07:45:50Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T03:33:19.225275Z","submitted_at":"2026-07-22T07:45:50Z","title":"Memory-Augmented Multimodal Large Language Models for Small Object Understanding in Streaming Aerial Videos"},"reference_resolution":{"displayed":47,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":47,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":47},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 47 of 47 outbound references and 0 inbound Pith citation observations for arXiv:2607.19857."}