{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:SEDGDZQIHFXZPWMPU7QYGPHOZE","short_pith_number":"pith:SEDGDZQI","schema_version":"1.0","canonical_sha256":"910661e608396f97d98fa7e1833ceec92218ca187f064b979916e939849eb935","source":{"kind":"arxiv","id":"2607.17386","version":1},"attestation_state":"computed","paper":{"title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyao Li, Chuang Zhang, Kaiwen Jing, Ming Wu, Ruixu Jia, Ruizhe Ou","submitted_at":"2026-07-19T19:07:19Z","abstract_excerpt":"Recent advances in Multimodal Large Language Models (MLLMs) have significantly improved remote sensing (RS) multimodal understanding. Language-conditioned segmentation is crucial for fine-grained target understanding in Unmanned Aerial Vehicle (UAV) videos. However, this task remains challenging due to the prevalence of small, visually ambiguous targets and dynamic aerial perspectives. In this paper, we propose SkyVLaM, a multimodal large language model for UAV video understanding. SkyVLaM constructs sparse tokens directly from patch-level video representations through a temporal basis perceiv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.17386","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-07-19T19:07:19Z","cross_cats_sorted":[],"title_canon_sha256":"c64edfcb4d10ee488e058a7b4d47407af9a91bf978d15afba45f05c5813d70f3","abstract_canon_sha256":"01e232fa5736a2ad6f356465f0e4ca3b944e58287ee16b950538a3f8411a508b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T01:21:30.547297Z","signature_b64":"y8kZb6z9r5+dF19iwxtCH3l+zNXAsMciRGaFewcf0Its0e3MRJGQlXttBK2zcWzq5JLBhChAoc7Wa3IaZ5ZOBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"910661e608396f97d98fa7e1833ceec92218ca187f064b979916e939849eb935","last_reissued_at":"2026-07-21T01:21:30.546368Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T01:21:30.546368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyao Li, Chuang Zhang, Kaiwen Jing, Ming Wu, Ruixu Jia, Ruizhe Ou","submitted_at":"2026-07-19T19:07:19Z","abstract_excerpt":"Recent advances in Multimodal Large Language Models (MLLMs) have significantly improved remote sensing (RS) multimodal understanding. Language-conditioned segmentation is crucial for fine-grained target understanding in Unmanned Aerial Vehicle (UAV) videos. However, this task remains challenging due to the prevalence of small, visually ambiguous targets and dynamic aerial perspectives. In this paper, we propose SkyVLaM, a multimodal large language model for UAV video understanding. SkyVLaM constructs sparse tokens directly from patch-level video representations through a temporal basis perceiv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.17386","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.17386/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.17386","created_at":"2026-07-21T01:21:30.546846+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.17386v1","created_at":"2026-07-21T01:21:30.546846+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.17386","created_at":"2026-07-21T01:21:30.546846+00:00"},{"alias_kind":"pith_short_12","alias_value":"SEDGDZQIHFXZ","created_at":"2026-07-21T01:21:30.546846+00:00"},{"alias_kind":"pith_short_16","alias_value":"SEDGDZQIHFXZPWMP","created_at":"2026-07-21T01:21:30.546846+00:00"},{"alias_kind":"pith_short_8","alias_value":"SEDGDZQI","created_at":"2026-07-21T01:21:30.546846+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE","json":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE.json","graph_json":"https://pith.science/api/pith-number/SEDGDZQIHFXZPWMPU7QYGPHOZE/graph.json","events_json":"https://pith.science/api/pith-number/SEDGDZQIHFXZPWMPU7QYGPHOZE/events.json","paper":"https://pith.science/paper/SEDGDZQI"},"agent_actions":{"view_html":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE","download_json":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE.json","view_paper":"https://pith.science/paper/SEDGDZQI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.17386&json=true","fetch_graph":"https://pith.science/api/pith-number/SEDGDZQIHFXZPWMPU7QYGPHOZE/graph.json","fetch_events":"https://pith.science/api/pith-number/SEDGDZQIHFXZPWMPU7QYGPHOZE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE/action/storage_attestation","attest_author":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE/action/author_attestation","sign_citation":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE/action/citation_signature","submit_replication":"https://pith.science/pith/SEDGDZQIHFXZPWMPU7QYGPHOZE/action/replication_record"}},"created_at":"2026-07-21T01:21:30.546846+00:00","updated_at":"2026-07-21T01:21:30.546846+00:00"}