{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:AX6YL2CL4HPNFVH4KK4PTTCSC6","short_pith_number":"pith:AX6YL2CL","schema_version":"1.0","canonical_sha256":"05fd85e84be1ded2d4fc52b8f9cc5217959cbb6fb76b64424d24c9b2f2fd1596","source":{"kind":"arxiv","id":"2607.24957","version":1},"attestation_state":"computed","paper":{"title":"PerceptionBench: Evaluating Atomic Visual Perception in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Qu, Chenzhuang Du, Haiming Wang, Haoning Wu, Haotian Yao, Hao Yang, Haoyu Lu, Hongcheng Gao, Jia Chen, Jia Li, Jinguo Zhu, Junwei Yang, Lin Sui, Mengfan Dong, Peizhou Cao, Tongtian Yue, Weihong Li, Xiaoxue Wu, Xinxing Zu, Xinyu Zhou, Yalin Wang, Yangyang Liu, Yao Wang, Y. Charles, Yifeng Xie, Yiping Bao, Yuhao Dong, Zaida Zhou, Zhangyang Qi, Zhiqi Huang, Zichao Lin, Zijia Zhao, Zuhao Yang","submitted_at":"2026-07-27T18:04:54Z","abstract_excerpt":"We introduce PerceptionBench, a benchmark specifically designed to evaluate the atomic visual perception capabilities of Multimodal Large Language Models (MLLMs). Existing benchmarks often fail to isolate perception: holistic evaluations conflate perceptual errors with failures in reasoning or domain knowledge, while application-driven benchmarks only cover narrow, fragmented domains shaped by heuristic designs. To address these limitations, PerceptionBench adopts a bottom-up approach: by diagnosing the earliest failure points in the responses of frontier MLLMs across 42 existing benchmarks, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.24957","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2026-07-27T18:04:54Z","cross_cats_sorted":[],"title_canon_sha256":"981acf00c051f33da0c93188c1ee8644755c2b1423c36f32140a2d6ef25870c8","abstract_canon_sha256":"dd25e2cc671908319506c279a40632fe7772a8258d4bf8fca32f499e829a2edd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-29T00:24:55.012942Z","signature_b64":"O0mNOGWA1qYkFgR7tlTJzJdG3kOSb9hhnI7qchv1CUzmsF2GSTzjUNmQ3CEImzBvWGqMD6AIYq21+bUEDWMWBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"05fd85e84be1ded2d4fc52b8f9cc5217959cbb6fb76b64424d24c9b2f2fd1596","last_reissued_at":"2026-07-29T00:24:55.012077Z","signature_status":"signed_v1","first_computed_at":"2026-07-29T00:24:55.012077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PerceptionBench: Evaluating Atomic Visual Perception in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Qu, Chenzhuang Du, Haiming Wang, Haoning Wu, Haotian Yao, Hao Yang, Haoyu Lu, Hongcheng Gao, Jia Chen, Jia Li, Jinguo Zhu, Junwei Yang, Lin Sui, Mengfan Dong, Peizhou Cao, Tongtian Yue, Weihong Li, Xiaoxue Wu, Xinxing Zu, Xinyu Zhou, Yalin Wang, Yangyang Liu, Yao Wang, Y. Charles, Yifeng Xie, Yiping Bao, Yuhao Dong, Zaida Zhou, Zhangyang Qi, Zhiqi Huang, Zichao Lin, Zijia Zhao, Zuhao Yang","submitted_at":"2026-07-27T18:04:54Z","abstract_excerpt":"We introduce PerceptionBench, a benchmark specifically designed to evaluate the atomic visual perception capabilities of Multimodal Large Language Models (MLLMs). Existing benchmarks often fail to isolate perception: holistic evaluations conflate perceptual errors with failures in reasoning or domain knowledge, while application-driven benchmarks only cover narrow, fragmented domains shaped by heuristic designs. To address these limitations, PerceptionBench adopts a bottom-up approach: by diagnosing the earliest failure points in the responses of frontier MLLMs across 42 existing benchmarks, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.24957","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.24957/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.24957","created_at":"2026-07-29T00:24:55.012514+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.24957v1","created_at":"2026-07-29T00:24:55.012514+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.24957","created_at":"2026-07-29T00:24:55.012514+00:00"},{"alias_kind":"pith_short_12","alias_value":"AX6YL2CL4HPN","created_at":"2026-07-29T00:24:55.012514+00:00"},{"alias_kind":"pith_short_16","alias_value":"AX6YL2CL4HPNFVH4","created_at":"2026-07-29T00:24:55.012514+00:00"},{"alias_kind":"pith_short_8","alias_value":"AX6YL2CL","created_at":"2026-07-29T00:24:55.012514+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6","json":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6.json","graph_json":"https://pith.science/api/pith-number/AX6YL2CL4HPNFVH4KK4PTTCSC6/graph.json","events_json":"https://pith.science/api/pith-number/AX6YL2CL4HPNFVH4KK4PTTCSC6/events.json","paper":"https://pith.science/paper/AX6YL2CL"},"agent_actions":{"view_html":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6","download_json":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6.json","view_paper":"https://pith.science/paper/AX6YL2CL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.24957&json=true","fetch_graph":"https://pith.science/api/pith-number/AX6YL2CL4HPNFVH4KK4PTTCSC6/graph.json","fetch_events":"https://pith.science/api/pith-number/AX6YL2CL4HPNFVH4KK4PTTCSC6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6/action/storage_attestation","attest_author":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6/action/author_attestation","sign_citation":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6/action/citation_signature","submit_replication":"https://pith.science/pith/AX6YL2CL4HPNFVH4KK4PTTCSC6/action/replication_record"}},"created_at":"2026-07-29T00:24:55.012514+00:00","updated_at":"2026-07-29T00:24:55.012514+00:00"}