{"as_of":"2026-08-09T09:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7a3a1208b99cb5f50f88b228d85329fbcaf52f3cd768b1451a70cd8ab98ea5f6","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T18:11:04.192167Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.17386/citation-record","integrity":"/paper/2607.17386/integrity","json":"/paper/2607.17386/citation-record.json","paper":"/paper/2607.17386"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.691639Z","title":"One token to seg them all: Language instructed reasoning seg- mentation in videos.Advances in Neural Information Processing Systems, 37:6833–6859, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.691639Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:559dfba864a89dc6e09c104b41f49521928ef2dfe7579a25ba2e6702d6354614","observation_id":"7c911503-5255-4be9-9c32-4a06b43b0c4a","resolution":{"observed_at":"2026-08-01T18:11:00.691639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.744246Z","title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.744246Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:cc0b7239d66555ee4052c679ac90be165bffb2c55f6fdbd3d1f1a364d6e2b8d3","observation_id":"e726ec18-5a18-4e01-a66d-acf19c22c573","resolution":{"observed_at":"2026-08-01T18:11:00.744246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.790110Z","title":"End-to-end referring video object segmentation with multimodal transformers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.790110Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:778a4210b1eb33453244f781586a78313c5a964df0fded66073020bda72082ee","observation_id":"bc7c76ca-1343-4e91-9975-31bbadbdd027","resolution":{"observed_at":"2026-08-01T18:11:00.790110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.854409Z","title":"Streamingtom: Streaming token compres- sion for efficient video understanding.arXiv preprint arXiv:2510.18269, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.854409Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:ec8c4a3d0caa88b780f9c1c154fa8ca0d35c446c5bbde7dd8df0d665d4a2c227","observation_id":"ebeeb569-da85-4acb-8d79-5b96c294852e","resolution":{"observed_at":"2026-08-01T18:11:00.854409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.919449Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.919449Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:0f3eb55d91759fb25127aec1d891e8f8e4b03b9dd8f16ef5198b533759fe756c","observation_id":"aac304ac-f50d-4d06-89f3-224d7201ef3f","resolution":{"observed_at":"2026-08-01T18:11:00.919449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.984738Z","title":"The unmanned aerial vehicle benchmark: Object detection and tracking","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.984738Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:7d9ca874c9783a3c1b40eff27bad466f7fa03f7c5ce2c4c06d0816ac36e7b663","observation_id":"f0495812-3451-4c19-85ed-ad679685c4c3","resolution":{"observed_at":"2026-08-01T18:11:00.984738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.082341Z","title":"Framefusion: Combining similarity and importance for video token reduction on large vision language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.082341Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:3b1a8eaba8422b74c7ab85de42167b288b82f9ed771f83961ed2a420d92a8c6f","observation_id":"f6ace37c-9484-43c0-8418-585971a9f62c","resolution":{"observed_at":"2026-08-01T18:11:01.082341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.00318","last_updated":"2026-04-04T04:38:43Z","snapshot_observed_at":"2026-07-06T21:34:05.672636Z","submitted_at":"2025-05-31T00:08:21Z","title":"Chain-of-Frames: Advancing Video Understanding in Multimodal LLMs via Frame-Aware Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.00318","snapshot_observed_at":"2026-08-01T18:11:01.204751Z","title":"Chain-of-frames: Advancing video understanding in multimodal llms via frame-aware reasoning.arXiv preprint arXiv:2506.00318, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.204751Z"},"links":{"cited_paper":"/paper/2506.00318","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:a646a9115397c1991ba8e333d37edc89844d827f835844ca32fbecc5fa514617","observation_id":"e7486384-4444-47a1-b64c-6ab04edcd40b","resolution":{"observed_at":"2026-08-01T18:11:01.204751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.296336Z","title":"The devil is in temporal token: High quality video reasoning segmentation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.296336Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:ea60008ea440e19654d74b81463511e4a32ae78db1693189659c69e8ec0ca980","observation_id":"954877f9-8f77-44ff-bdd2-48c34a27fa1d","resolution":{"observed_at":"2026-08-01T18:11:01.296336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.367418Z","title":"Rsgpt: A remote sensing vision language model and benchmark.ISPRS Journal of Photogrammetry and Remote Sensing, 224:272–286, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.367418Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:4a80b92abfddc465dee837159b65a95d068298a966dae7fc536379aca6852c1f","observation_id":"c2dae05a-a68c-4d7b-9543-b872f27f70de","resolution":{"observed_at":"2026-08-01T18:11:01.367418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.457054Z","title":"Prunevid: Visual token pruning for efficient video large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.457054Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:b049f73118ca8dcd91c569b012fdf3e020b3482b595bd264c7a0c70a45298939","observation_id":"69ff8e28-3474-4e55-8196-1d2bf37d04b2","resolution":{"observed_at":"2026-08-01T18:11:01.457054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.521166Z","title":"Multi-granular spatio-temporal token merging for training-free acceleration of video llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.521166Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:23c4592fe765ea045d7f02c27557aa47009a41a87fbc2a9fb6123f54506686af","observation_id":"dfecb740-b81a-49cf-a800-cfeac3bca07f","resolution":{"observed_at":"2026-08-01T18:11:01.521166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06234","last_updated":"2025-01-27T01:45:15Z","snapshot_observed_at":"2026-08-08T12:33:45.268848Z","submitted_at":"2024-10-08T17:45:51Z","title":"TEOChat: A Large Vision-Language Assistant for Temporal Earth Observation Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06234","snapshot_observed_at":"2026-08-01T18:11:01.580630Z","title":"Teochat: A large vision-language assistant for temporal earth observation data.arXiv preprint arXiv:2410.06234, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.580630Z"},"links":{"cited_paper":"/paper/2410.06234","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:6d694265ff7f00bfb973372478119bed1fd4dd21dfa0d00d3b2efc276451e0ad","observation_id":"c57608d3-d131-4995-8b1c-eff11246edac","resolution":{"observed_at":"2026-08-01T18:11:01.580630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.696273Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.696273Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:05e8b38938a2055b4fd1b1b53b2680dd150a9434c3d09a6d6b9d725fa059c410","observation_id":"a485416f-d454-4b76-9ef5-7fc4ede0b8cf","resolution":{"observed_at":"2026-08-01T18:11:01.696273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.761048Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.761048Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:ad16651272b8d159801fe4db5824ccd58d8f1a6a9fd5bdbfc0bd1a2e1c82f692","observation_id":"83cd5b5c-dd0b-47d7-a5e5-ee9ef889736f","resolution":{"observed_at":"2026-08-01T18:11:01.761048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.849069Z","title":"Referdino: Referring video object segmentation with visual grounding foundations","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.849069Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:693e51ad49e0fe92dfff56d63cf4dd7a4026d674b2a49b04d83bed055dc085cc","observation_id":"6731052e-dbfa-41ad-814f-7b040ed21374","resolution":{"observed_at":"2026-08-01T18:11:01.849069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.971650Z","title":"Video-llava: Learning united visual representation by alignment before projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.971650Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:8b8033ed5140d2a306e37002969400c0d1bc1149443d2a5c5638dcf8715386d7","observation_id":"4f45cead-57fa-4193-934d-3aad75ef807a","resolution":{"observed_at":"2026-08-01T18:11:01.971650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.034375Z","title":"Glus: Global-local reasoning unified into a single large language model for video segmentation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.034375Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:151306b2378e4228e842f795555a78f0bbe76583a4fded9527687ae4ffe2c0a2","observation_id":"27c684a3-7358-48c6-886a-e007211e6294","resolution":{"observed_at":"2026-08-01T18:11:02.034375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.108301Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36:34892–34916, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.108301Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:31b9822a64bd95e17b381ac7c3239bfae7b71df25e7d9a7dbb23a44c7d239e54","observation_id":"1e1e3f91-2614-4465-aaeb-bde5dc3df7a4","resolution":{"observed_at":"2026-08-01T18:11:02.108301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05679","last_updated":"2024-12-10T02:23:30Z","snapshot_observed_at":"2026-07-06T20:03:16.082369Z","submitted_at":"2024-12-07T15:11:21Z","title":"RSUniVLM: A Unified Vision Language Model for Remote Sensing via Granularity-oriented Mixture of Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05679","snapshot_observed_at":"2026-08-01T18:11:02.265647Z","title":"Rsunivlm: A unified vision language model for remote sensing via granularity-oriented mixture of experts.arXiv preprint arXiv:2412.05679, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.265647Z"},"links":{"cited_paper":"/paper/2412.05679","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:2d56de3070eb7401949c60a755a41426eae9075cb49a246a8b2b92062078e983","observation_id":"bf66e690-c950-4d14-b84c-12030a074936","resolution":{"observed_at":"2026-08-01T18:11:02.265647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10100","last_updated":"2024-07-08T04:33:37Z","snapshot_observed_at":"2026-07-06T18:31:04.425917Z","submitted_at":"2024-06-14T14:57:07Z","title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10100","snapshot_observed_at":"2026-08-01T18:11:02.304478Z","title":"Skysensegpt: A fine-grained instruction tuning dataset and model for remote sensing vision-language understanding.arXiv preprint arXiv:2406.10100, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.304478Z"},"links":{"cited_paper":"/paper/2406.10100","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:761bb5b052cd06c661b616c510c45ab168e120747f3018e576cc6712f4494e8c","observation_id":"7da24c0e-c78c-49c0-915a-672bacdb4bf6","resolution":{"observed_at":"2026-08-01T18:11:02.304478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.369009Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.369009Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:a39b144f8c44b1cf8ef49468e5a5b2bdd1c23235e9acc9d9b705da4159b20dfb","observation_id":"0ebe0656-7020-49d2-88e4-e00b58c04323","resolution":{"observed_at":"2026-08-01T18:11:02.369009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.428042Z","title":"Videoglamm: A large multimodal model for pixel-level visual grounding in videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.428042Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:44bdf3dd48bdb85ef8dda3bd4df728e14bedcf59e66d831ff288d3644df32af2","observation_id":"b7b0d143-8b00-4c13-8bfd-e131ad13230a","resolution":{"observed_at":"2026-08-01T18:11:02.428042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.501383Z","title":"Geopix: A multimodal large language model for pixel-level image understanding in remote sensing.IEEE Geoscience and Remote Sensing Magazine, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.501383Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:6b2f482027b56e03a9bee7f3c2392f7b543e15148215161ba79fcdd8cb2d1a46","observation_id":"113d5682-d67d-40cb-b88c-a2e3a48dad5a","resolution":{"observed_at":"2026-08-01T18:11:02.501383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.552720Z","title":"Vhm: Versatile and honest vision language model for remote sensing image analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.552720Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:4310dade618d80c9f93a260d3dba570b2cb862606750d1b1b3b01a2f50280c82","observation_id":"65940e87-c7ca-4c12-b52c-4d4054028358","resolution":{"observed_at":"2026-08-01T18:11:02.552720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.593053Z","title":"Llava++: extending visual capabilities with llama-3 and phi-3 (2024).URL https://github.com/mbzuai-oryx/LLaVA- pp, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.593053Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:a7190c4bec6c5df9a87ee70db9070213b667dcab2961cbde5df285a5727372a5","observation_id":"9ddf7593-9019-48ce-b84c-fa5f2f6664a9","resolution":{"observed_at":"2026-08-01T18:11:02.593053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.644608Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.644608Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:f9ff9c6a4d9e4a2513ed0ca32bf22d9f68647b64c961122e26ef7f7601681a6a","observation_id":"086ae9bd-1812-4e22-acf0-874e4e16621b","resolution":{"observed_at":"2026-08-01T18:11:02.644608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-01T18:11:02.721970Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.721970Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:cf597c4ce10822ac6c936db19f28aa369200495bcb9e6efbe859fd808e1558c0","observation_id":"fb476229-664f-4803-89e2-2fd277f44817","resolution":{"observed_at":"2026-08-01T18:11:02.721970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.803839Z","title":"Pixellm: Pixel reasoning with large multimodal model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.803839Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:b3ed360f47678a858f8c356de9acb21795253a9eaad86e2ec2e10472460e3478","observation_id":"eec54fbe-cb03-45c0-b031-d8fcb329e4d4","resolution":{"observed_at":"2026-08-01T18:11:02.803839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.860093Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.860093Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:6591295478b26ddfc697652e5f389fdedcd9bb3c6b0b44d3f484919f2c4b75ff","observation_id":"867c4e28-917a-4a41-93be-31ea3459c95a","resolution":{"observed_at":"2026-08-01T18:11:02.860093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.915593Z","title":"Earthdial: Turning multi-sensory earth observations to interactive dialogues","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.915593Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:319cc80866720cca85a743e40b7f6b9d1d178bdcaf399d0364e2c86961e677c1","observation_id":"d83d7eb4-6c63-4448-94a0-952dadbdb274","resolution":{"observed_at":"2026-08-01T18:11:02.915593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.973608Z","title":"Drone-based rgb-infrared cross-modality vehicle detection via uncertainty-aware learning.IEEE Transactions on Circuits and Systems for Video Technology, 32(10):6700–6713, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.973608Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:2e3961e024661a31540a358e3660d536c4096e08b93dd611a7815b2b78613b3f","observation_id":"52daab6d-b37e-48b3-9d2e-d8222f199e69","resolution":{"observed_at":"2026-08-01T18:11:02.973608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.030943Z","title":"Adaptive keyframe sampling for long video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.030943Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:1e1b03031879c06e2a6fd85329bc694be270b2e9ee973fccbf565fd31ab06e9e","observation_id":"17863fb7-2119-4768-8b96-53666e87d913","resolution":{"observed_at":"2026-08-01T18:11:03.030943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.074265Z","title":"Cider: Consensus-based image description evaluation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.074265Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:6f08436c71baa93f3da8dbd76aedf18d7b56a346fdda23b3605d27c5eb1b862f","observation_id":"4e899a43-c9ad-4aeb-9ac4-0626688eb142","resolution":{"observed_at":"2026-08-01T18:11:03.074265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.124523Z","title":"Geollava-8k: scaling remote-sensing multimodal large language models to 8k resolution.arXiv preprint arXiv:2505.21375, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.124523Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:24876ac0e76f201f923ece73e159e24cec924fc90d97638721e4a740f8070655","observation_id":"a24a7245-aef3-4bb0-aa6b-10394c3e1801","resolution":{"observed_at":"2026-08-01T18:11:03.124523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.221579Z","title":"Instructseg: Unifying instructed visual segmentation with multi-modal large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.221579Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:d840692f02ae2bcefa16038f47744ec8192108d821717d206083a0add9d512f3","observation_id":"2dd4ab56-621a-491c-bbf2-07dda40d7475","resolution":{"observed_at":"2026-08-01T18:11:03.221579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.350493Z","title":"Longvlm: Efficient long video understanding via large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.350493Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:6eacfd0e43f9f6d0ac8d8a5c0360935e2a27708dd3e40148b8641bb60a446ffa","observation_id":"aa94ce04-88b5-493b-9d0e-b29f1600024d","resolution":{"observed_at":"2026-08-01T18:11:03.350493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.422889Z","title":"Language as queries for referring video object segmentation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.422889Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:c216041cc66f006a3ad98f86ceb6885dd96dfa0f588f7da747983fcd36132bc4","observation_id":"1a711df3-dda3-4d74-9bfd-c7b3a4daab95","resolution":{"observed_at":"2026-08-01T18:11:03.422889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-08-08T03:13:28.000968Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-01T18:11:03.486399Z","title":"Slowfast-llava: A strong training-free baseline for video large language models.arXiv preprint arXiv:2407.15841, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.486399Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:0ae4f1e1afa5bd0337dce25ba6bd2a6955c78120fdba1ee13384390a8c1cd671","observation_id":"9ce06e3c-e938-4b4b-ab75-acf4c575595c","resolution":{"observed_at":"2026-08-01T18:11:03.486399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.576528Z","title":"Visa: Reasoning video object segmentation via large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.576528Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:c05a7e591359a09978079b6cf7fbf6f975c94317c9abda5a800d6027dfa9efe9","observation_id":"32849a1b-d5f7-4a20-b197-fc29806e3be2","resolution":{"observed_at":"2026-08-01T18:11:03.576528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.579616Z","title":"Referred by multi-modality: A unified temporal transformer for video object segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.579616Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:6b849a13abe5aa9c8dd54021aab6448045e03446f090cadce86dc601cc5532d7","observation_id":"126070c8-ffb6-4120-bcfd-8e74f9716739","resolution":{"observed_at":"2026-08-01T18:11:03.579616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.612401Z","title":"Self-chained image-language model for video localization and question answering.Advances in Neural Information Processing Systems, 36:76749–76771, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.612401Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:8fb27872574138f140c818e01c1aa8b8900983324a8733150eb686790a90b4a4","observation_id":"f4431f6f-4b71-442c-8f56-39b9b7774506","resolution":{"observed_at":"2026-08-01T18:11:03.612401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03226","last_updated":"2025-03-28T03:19:52Z","snapshot_observed_at":"2026-07-06T19:27:33.934338Z","submitted_at":"2024-10-04T08:26:06Z","title":"Frame-Voyager: Learning to Query Frames for Video Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03226","snapshot_observed_at":"2026-08-01T18:11:03.661242Z","title":"Frame-voyager: Learning to query frames for video large language models.arXiv preprint arXiv:2410.03226, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.661242Z"},"links":{"cited_paper":"/paper/2410.03226","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:4689590de02132076597ae670b474b0772016a328ab1e7a4aa58a8d9c00ab9de","observation_id":"4a2ee8db-e65b-4904-8f56-7f723e0d407d","resolution":{"observed_at":"2026-08-01T18:11:03.661242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-08T01:58:42.644918Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-08-01T18:11:03.743258Z","title":"Sa2va: Marrying sam2 with llava for dense grounded understanding of images and videos.arXiv preprint arXiv:2501.04001, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.743258Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:b26d03a3091e1d086cc9c39eaf8d91dbacd359c9bcf4275a8a429a9420e91567","observation_id":"4817844c-a271-4236-a712-ef3db908dedc","resolution":{"observed_at":"2026-08-01T18:11:03.743258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.814064Z","title":"Skyeyegpt: Unifying remote sensing vision- language tasks via instruction tuning with large language model.ISPRS Journal of Photogram- metry and Remote Sensing, 221:64–77, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.814064Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:b594a192da69bf145d9c6d70ba5b6458acb5769e218bf5d951138a8bac450628","observation_id":"bb233953-67a4-4e45-9c3c-ed97252e8ef0","resolution":{"observed_at":"2026-08-01T18:11:03.814064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.885115Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.885115Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:0a7fd5853203d519958259eff1cab1332e8a5371860b46c0083b9b5dc5def3bf","observation_id":"bd172da8-7b7c-4156-9dcf-d6053158fda9","resolution":{"observed_at":"2026-08-01T18:11:03.885115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12490","last_updated":"2025-03-16T12:48:17Z","snapshot_observed_at":"2026-08-07T17:00:38.055868Z","submitted_at":"2025-03-16T12:48:17Z","title":"GeoRSMLLM: A Multimodal Large Language Model for Vision-Language Tasks in Geoscience and Remote Sensing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.12490","snapshot_observed_at":"2026-08-01T18:11:03.944636Z","title":"Georsmllm: A multimodal large language model for vision-language tasks in geoscience and remote sensing.arXiv preprint arXiv:2503.12490, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.944636Z"},"links":{"cited_paper":"/paper/2503.12490","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:5b2a8a2b7ad62022424e29ed7642623d29ad1d6fa3caf04ec630fd6a82c8fe55","observation_id":"0cd98b9a-d26e-4ea7-90ad-819a85d924ee","resolution":{"observed_at":"2026-08-01T18:11:03.944636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.007847Z","title":"Tifre: Text-guided video frame reduction for efficient video multi-modal large language models.arXiv preprint arXiv:2602.08861, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.007847Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:6356fd78eb8bb388293765fad2e64729040c9fef3c2636ca62f2af51cc40a18d","observation_id":"a1b883f2-6312-4707-a5a4-da08e0549566","resolution":{"observed_at":"2026-08-01T18:11:04.007847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.079355Z","title":"Reason: Reinforced causal search with information bottleneck for video understanding","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.079355Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:db42f9fe6feba6d864cb01767fea9956f51cf59ac61d1f16eb5938bd5ebeb23c","observation_id":"9860a9dd-5836-408c-b37e-fe9a0a0d5ded","resolution":{"observed_at":"2026-08-01T18:11:04.079355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.130946Z","title":"Detection and tracking meet drones challenge.IEEE transactions on pattern analysis and machine intelligence, 44(11):7380–7399, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.130946Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:860940f70a73bda3dd76218699b7f28c7ee14c9c8c843318062bb6f05f18e8ad","observation_id":"890e5b91-6ce5-49a7-9e45-11915eb46598","resolution":{"observed_at":"2026-08-01T18:11:04.130946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.192167Z","title":"Focus: Efficient keyframe selection for long video understanding.arXiv preprint arXiv:2510.27280, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.192167Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:b0510b8f98ddc3cbbb4cde75dfcd6ac657f6a7853f547c864406c10fcaf1a819","observation_id":"42c5d8b8-44ec-4295-ac63-564a1914705d","resolution":{"observed_at":"2026-08-01T18:11:04.192167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T19:25:49.248646Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":51,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 0 inbound Pith citation observations for arXiv:2607.17386."}