{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QEREYWCSR3OUYBNVO34L4CMTCZ","short_pith_number":"pith:QEREYWCS","schema_version":"1.0","canonical_sha256":"81224c58528edd4c05b576f8be099316548638c37158bc14513ad7fbc1518d8b","source":{"kind":"arxiv","id":"2505.12207","version":3},"attestation_state":"computed","paper":{"title":"Can Large Multimodal Models Understand Agricultural Scenes? Benchmarking with AgroMind","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Haohuan Fu, Henglian Huang, Jianxi Huang, Jiarui Zhang, Juepeng Zheng, Qingmei Li, Shuohong Lou, Weijia Li, Yang Zhang, Yibin Wen, Yuhang Chen, Zhiwei Zhang, Zurong Mai","submitted_at":"2025-05-18T02:45:19Z","abstract_excerpt":"Large Multimodal Models (LMMs) has demonstrated capabilities across various domains, but comprehensive benchmarks for agricultural remote sensing (RS) remain scarce. Existing benchmarks designed for agricultural RS scenarios exhibit notable limitations, primarily in terms of insufficient scene diversity in the dataset and oversimplified task design. To bridge this gap, we introduce AgroMind, a comprehensive agricultural remote sensing benchmark covering four task dimensions: spatial perception, object understanding, scene understanding, and scene reasoning, with a total of 13 task types, rangi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12207","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-18T02:45:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"af2230de5c523e71ac59116fb2bd447abe6bbc13cb1e11f96d11ba477f4b602a","abstract_canon_sha256":"b7ac5c181a11e5041526301afef06bdc39cb317c3ef037084008102962bab060"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:03.032013Z","signature_b64":"KLHw1lkFf924gCSJAJMVTeAHN8kvWQhKscyIebB+OhLZJ7IfDAupWqULpJEBxzJAPNTwqVIruaaRy1APy6MUAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81224c58528edd4c05b576f8be099316548638c37158bc14513ad7fbc1518d8b","last_reissued_at":"2026-07-05T11:53:03.031567Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:03.031567Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Multimodal Models Understand Agricultural Scenes? Benchmarking with AgroMind","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Haohuan Fu, Henglian Huang, Jianxi Huang, Jiarui Zhang, Juepeng Zheng, Qingmei Li, Shuohong Lou, Weijia Li, Yang Zhang, Yibin Wen, Yuhang Chen, Zhiwei Zhang, Zurong Mai","submitted_at":"2025-05-18T02:45:19Z","abstract_excerpt":"Large Multimodal Models (LMMs) has demonstrated capabilities across various domains, but comprehensive benchmarks for agricultural remote sensing (RS) remain scarce. Existing benchmarks designed for agricultural RS scenarios exhibit notable limitations, primarily in terms of insufficient scene diversity in the dataset and oversimplified task design. To bridge this gap, we introduce AgroMind, a comprehensive agricultural remote sensing benchmark covering four task dimensions: spatial perception, object understanding, scene understanding, and scene reasoning, with a total of 13 task types, rangi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12207","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12207","created_at":"2026-07-05T11:53:03.031624+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12207v3","created_at":"2026-07-05T11:53:03.031624+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12207","created_at":"2026-07-05T11:53:03.031624+00:00"},{"alias_kind":"pith_short_12","alias_value":"QEREYWCSR3OU","created_at":"2026-07-05T11:53:03.031624+00:00"},{"alias_kind":"pith_short_16","alias_value":"QEREYWCSR3OUYBNV","created_at":"2026-07-05T11:53:03.031624+00:00"},{"alias_kind":"pith_short_8","alias_value":"QEREYWCS","created_at":"2026-07-05T11:53:03.031624+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25784","citing_title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22366","citing_title":"AgroTools: A Benchmark for Tool-Augmented Multimodal Agents in Agriculture","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2511.23253","citing_title":"AgroCoT: A Chain-of-Thought Benchmark for Evaluating Reasoning in Vision-Language Models for Agriculture","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08884","citing_title":"HM-Bench: A Comprehensive Benchmark for Multimodal Large Language Models in Hyperspectral Remote Sensing","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ","json":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ.json","graph_json":"https://pith.science/api/pith-number/QEREYWCSR3OUYBNVO34L4CMTCZ/graph.json","events_json":"https://pith.science/api/pith-number/QEREYWCSR3OUYBNVO34L4CMTCZ/events.json","paper":"https://pith.science/paper/QEREYWCS"},"agent_actions":{"view_html":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ","download_json":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ.json","view_paper":"https://pith.science/paper/QEREYWCS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12207&json=true","fetch_graph":"https://pith.science/api/pith-number/QEREYWCSR3OUYBNVO34L4CMTCZ/graph.json","fetch_events":"https://pith.science/api/pith-number/QEREYWCSR3OUYBNVO34L4CMTCZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ/action/storage_attestation","attest_author":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ/action/author_attestation","sign_citation":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ/action/citation_signature","submit_replication":"https://pith.science/pith/QEREYWCSR3OUYBNVO34L4CMTCZ/action/replication_record"}},"created_at":"2026-07-05T11:53:03.031624+00:00","updated_at":"2026-07-05T11:53:03.031624+00:00"}