{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:NLZY6X4TQQJDB6ZEOETL6ZFLTN","short_pith_number":"pith:NLZY6X4T","schema_version":"1.0","canonical_sha256":"6af38f5f93841230fb247126bf64ab9b68b47ae222a3007ff11c97cce0097bf4","source":{"kind":"arxiv","id":"2607.22014","version":1},"attestation_state":"computed","paper":{"title":"Zero-Shot Mission-Level Evaluation for Aerial MLLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.RO"],"primary_cat":"cs.AI","authors_text":"Ishaan Bhimwal, Jona Ruthardt, Ryousuke Yamada, Suman Navaratnarajah, Taehyoung Kim, Wolfram Burgard, Yannik Blei, Yuki M Asano","submitted_at":"2026-07-24T06:22:50Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are emerging as core reasoning modules for embodied agents, yet it remains unclear how well general-purpose models can solve long-horizon embodied tasks from a single high-level instruction. We introduce MissionBench, a benchmark for mission-level evaluation of MLLMs in aerial 3D environments. It comprises 120 missions across five simulated 3D environments and four task families. Agents must autonomously plan, navigate, and report outcomes using only egocentric observations and its action history, without aerial-specific fine-tuning. Across 22 open- and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.22014","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-24T06:22:50Z","cross_cats_sorted":["cs.CL","cs.CV","cs.RO"],"title_canon_sha256":"d94745dd7e87b40b06000303b1a1e625ad28d53d2f739a5ae210da12a8dea826","abstract_canon_sha256":"e0a54c4b5fc2282d29a0cf7f0602c4dc2c75674e3e6a15fe84856b918e80eeb6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-27T01:20:43.231167Z","signature_b64":"ZRcZCi+vyJp096wCWbxpwId5YYyq0mPQ5SA057nnC52usrSgZZzaZy54Ojmm/5AEAC0ouAeuz94/8J4mJjszDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6af38f5f93841230fb247126bf64ab9b68b47ae222a3007ff11c97cce0097bf4","last_reissued_at":"2026-07-27T01:20:43.230371Z","signature_status":"signed_v1","first_computed_at":"2026-07-27T01:20:43.230371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero-Shot Mission-Level Evaluation for Aerial MLLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.RO"],"primary_cat":"cs.AI","authors_text":"Ishaan Bhimwal, Jona Ruthardt, Ryousuke Yamada, Suman Navaratnarajah, Taehyoung Kim, Wolfram Burgard, Yannik Blei, Yuki M Asano","submitted_at":"2026-07-24T06:22:50Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are emerging as core reasoning modules for embodied agents, yet it remains unclear how well general-purpose models can solve long-horizon embodied tasks from a single high-level instruction. We introduce MissionBench, a benchmark for mission-level evaluation of MLLMs in aerial 3D environments. It comprises 120 missions across five simulated 3D environments and four task families. Agents must autonomously plan, navigate, and report outcomes using only egocentric observations and its action history, without aerial-specific fine-tuning. Across 22 open- and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.22014","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.22014/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.22014","created_at":"2026-07-27T01:20:43.230784+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.22014v1","created_at":"2026-07-27T01:20:43.230784+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.22014","created_at":"2026-07-27T01:20:43.230784+00:00"},{"alias_kind":"pith_short_12","alias_value":"NLZY6X4TQQJD","created_at":"2026-07-27T01:20:43.230784+00:00"},{"alias_kind":"pith_short_16","alias_value":"NLZY6X4TQQJDB6ZE","created_at":"2026-07-27T01:20:43.230784+00:00"},{"alias_kind":"pith_short_8","alias_value":"NLZY6X4T","created_at":"2026-07-27T01:20:43.230784+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN","json":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN.json","graph_json":"https://pith.science/api/pith-number/NLZY6X4TQQJDB6ZEOETL6ZFLTN/graph.json","events_json":"https://pith.science/api/pith-number/NLZY6X4TQQJDB6ZEOETL6ZFLTN/events.json","paper":"https://pith.science/paper/NLZY6X4T"},"agent_actions":{"view_html":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN","download_json":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN.json","view_paper":"https://pith.science/paper/NLZY6X4T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.22014&json=true","fetch_graph":"https://pith.science/api/pith-number/NLZY6X4TQQJDB6ZEOETL6ZFLTN/graph.json","fetch_events":"https://pith.science/api/pith-number/NLZY6X4TQQJDB6ZEOETL6ZFLTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN/action/storage_attestation","attest_author":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN/action/author_attestation","sign_citation":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN/action/citation_signature","submit_replication":"https://pith.science/pith/NLZY6X4TQQJDB6ZEOETL6ZFLTN/action/replication_record"}},"created_at":"2026-07-27T01:20:43.230784+00:00","updated_at":"2026-07-27T01:20:43.230784+00:00"}