{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:ILHECK6TT5SKDFXMCCHHFJZHW6","short_pith_number":"pith:ILHECK6T","schema_version":"1.0","canonical_sha256":"42ce412bd39f64a196ec108e72a727b78dc410448eb8c19f1c2fc2d84d628b3c","source":{"kind":"arxiv","id":"2608.11738","version":1},"attestation_state":"computed","paper":{"title":"Advancing MLLM-based UAV Image Understanding and Reasoning: A Benchmark and a Training-Free Multi-Agent System","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Haoyu Zhang, Jiakang Yuan, Lin Zhang, Peng Ye, Shenghong Yi, Shuoxun Zhang, Tao Chen, Yuening Wang","submitted_at":"2026-08-12T07:25:32Z","abstract_excerpt":"Multimodal Large Language Model (MLLM)-based UAV aerial image understanding and reasoning is essential for aerial intelligence yet poses distinct challenges arising from extreme scale variation, arbitrary camera orientations, and high object density. Despite growing interest, existing evaluations remain fragmented across individual datasets and narrow tasks, leaving a critical gap in unified assessment of UAV understanding and reasoning capabilities. To fill this gap, we construct UAVQA-Bench, a benchmark of 1,500 human-annotated QA pairs drawn from 13 public UAV datasets, covering 6 capabilit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.11738","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-08-12T07:25:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1e119b99b8cca974e27e57ae9baabc9b6db2cd48fff1b07bc6d4cf6c23b9bbf3","abstract_canon_sha256":"ba5919b64b5e951eb4e6630e50c28712470691e0c3c28de656e3ad3b0fe07798"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-13T01:27:58.349046Z","signature_b64":"vA6rTeuWN/rzkm6aB1GFX8llQcYapneMD80dYsXS2o+jj5maRKiGhCmlEHcX7yZgiJmZzpm5Cav7mJ0bDv54Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"42ce412bd39f64a196ec108e72a727b78dc410448eb8c19f1c2fc2d84d628b3c","last_reissued_at":"2026-08-13T01:27:58.346871Z","signature_status":"signed_v1","first_computed_at":"2026-08-13T01:27:58.346871Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Advancing MLLM-based UAV Image Understanding and Reasoning: A Benchmark and a Training-Free Multi-Agent System","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Haoyu Zhang, Jiakang Yuan, Lin Zhang, Peng Ye, Shenghong Yi, Shuoxun Zhang, Tao Chen, Yuening Wang","submitted_at":"2026-08-12T07:25:32Z","abstract_excerpt":"Multimodal Large Language Model (MLLM)-based UAV aerial image understanding and reasoning is essential for aerial intelligence yet poses distinct challenges arising from extreme scale variation, arbitrary camera orientations, and high object density. Despite growing interest, existing evaluations remain fragmented across individual datasets and narrow tasks, leaving a critical gap in unified assessment of UAV understanding and reasoning capabilities. To fill this gap, we construct UAVQA-Bench, a benchmark of 1,500 human-annotated QA pairs drawn from 13 public UAV datasets, covering 6 capabilit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.11738","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.11738/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.11738","created_at":"2026-08-13T01:27:58.347998+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.11738v1","created_at":"2026-08-13T01:27:58.347998+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.11738","created_at":"2026-08-13T01:27:58.347998+00:00"},{"alias_kind":"pith_short_12","alias_value":"ILHECK6TT5SK","created_at":"2026-08-13T01:27:58.347998+00:00"},{"alias_kind":"pith_short_16","alias_value":"ILHECK6TT5SKDFXM","created_at":"2026-08-13T01:27:58.347998+00:00"},{"alias_kind":"pith_short_8","alias_value":"ILHECK6T","created_at":"2026-08-13T01:27:58.347998+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6","json":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6.json","graph_json":"https://pith.science/api/pith-number/ILHECK6TT5SKDFXMCCHHFJZHW6/graph.json","events_json":"https://pith.science/api/pith-number/ILHECK6TT5SKDFXMCCHHFJZHW6/events.json","paper":"https://pith.science/paper/ILHECK6T"},"agent_actions":{"view_html":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6","download_json":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6.json","view_paper":"https://pith.science/paper/ILHECK6T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.11738&json=true","fetch_graph":"https://pith.science/api/pith-number/ILHECK6TT5SKDFXMCCHHFJZHW6/graph.json","fetch_events":"https://pith.science/api/pith-number/ILHECK6TT5SKDFXMCCHHFJZHW6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6/action/storage_attestation","attest_author":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6/action/author_attestation","sign_citation":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6/action/citation_signature","submit_replication":"https://pith.science/pith/ILHECK6TT5SKDFXMCCHHFJZHW6/action/replication_record"}},"created_at":"2026-08-13T01:27:58.347998+00:00","updated_at":"2026-08-13T01:27:58.347998+00:00"}