{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q6CGFPVRWDFY2YSY24E7ZKAKZ4","short_pith_number":"pith:Q6CGFPVR","schema_version":"1.0","canonical_sha256":"878462beb1b0cb8d6258d709fca80acf03b4c84eb094357e31257275814a58a8","source":{"kind":"arxiv","id":"2505.08765","version":2},"attestation_state":"computed","paper":{"title":"Towards Autonomous UAV Visual Object Search in City Space: Benchmark and Agentic Methodology","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Beidan Liu, Chen Gao, Quanjun Yin, Sihang Qiu, Yatai Ji, Yihao Zhao, Yong Li, Yong Zhao, Yue Hu, Zhengqiu Zhu","submitted_at":"2025-05-13T17:34:54Z","abstract_excerpt":"Aerial Visual Object Search (AVOS) tasks in urban environments require Unmanned Aerial Vehicles (UAVs) to autonomously search for and identify target objects using visual and textual cues without external guidance. Existing approaches struggle in complex urban environments due to redundant semantic processing, similar object distinction, and the exploration-exploitation dilemma. To bridge this gap and support the AVOS task, we introduce CityAVOS, the first benchmark dataset for autonomous search of common urban objects. This dataset comprises 2,420 tasks across six object categories with varyi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08765","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-13T17:34:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"97d2a24accf86a5f64b95c37b4372373e96f995e04dcb042a6fb4dcba9a91149","abstract_canon_sha256":"7532a1525f72e6b8ba88b0d255c7ed9c8dedf50a1dec432a7f5bce9b60bac869"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:44.281739Z","signature_b64":"Ntpb/ukesYx13IrYTa0hFVwH5hsx+ut3ouIfVSIuF0nPmE3uMEPYkakddzt9rYMq1PDX4Mo7lR6H6+y2fvV/CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"878462beb1b0cb8d6258d709fca80acf03b4c84eb094357e31257275814a58a8","last_reissued_at":"2026-07-05T11:02:44.281219Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:44.281219Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Autonomous UAV Visual Object Search in City Space: Benchmark and Agentic Methodology","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Beidan Liu, Chen Gao, Quanjun Yin, Sihang Qiu, Yatai Ji, Yihao Zhao, Yong Li, Yong Zhao, Yue Hu, Zhengqiu Zhu","submitted_at":"2025-05-13T17:34:54Z","abstract_excerpt":"Aerial Visual Object Search (AVOS) tasks in urban environments require Unmanned Aerial Vehicles (UAVs) to autonomously search for and identify target objects using visual and textual cues without external guidance. Existing approaches struggle in complex urban environments due to redundant semantic processing, similar object distinction, and the exploration-exploitation dilemma. To bridge this gap and support the AVOS task, we introduce CityAVOS, the first benchmark dataset for autonomous search of common urban objects. This dataset comprises 2,420 tasks across six object categories with varyi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08765","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08765/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08765","created_at":"2026-07-05T11:02:44.281289+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08765v2","created_at":"2026-07-05T11:02:44.281289+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08765","created_at":"2026-07-05T11:02:44.281289+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q6CGFPVRWDFY","created_at":"2026-07-05T11:02:44.281289+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q6CGFPVRWDFY2YSY","created_at":"2026-07-05T11:02:44.281289+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q6CGFPVR","created_at":"2026-07-05T11:02:44.281289+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07973","citing_title":"How Far Are Large Multimodal Models from Human-Level Spatial Action? A Benchmark for Goal-Oriented Embodied Navigation in Urban Airspace","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4","json":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4.json","graph_json":"https://pith.science/api/pith-number/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/graph.json","events_json":"https://pith.science/api/pith-number/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/events.json","paper":"https://pith.science/paper/Q6CGFPVR"},"agent_actions":{"view_html":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4","download_json":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4.json","view_paper":"https://pith.science/paper/Q6CGFPVR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08765&json=true","fetch_graph":"https://pith.science/api/pith-number/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/graph.json","fetch_events":"https://pith.science/api/pith-number/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/action/storage_attestation","attest_author":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/action/author_attestation","sign_citation":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/action/citation_signature","submit_replication":"https://pith.science/pith/Q6CGFPVRWDFY2YSY24E7ZKAKZ4/action/replication_record"}},"created_at":"2026-07-05T11:02:44.281289+00:00","updated_at":"2026-07-05T11:02:44.281289+00:00"}