{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FT4INJ2AGM2WTKQRDTKZBIA3FB","short_pith_number":"pith:FT4INJ2A","schema_version":"1.0","canonical_sha256":"2cf886a740333569aa111cd590a01b2851c24e15d76644eb53d0ac6e9db460b7","source":{"kind":"arxiv","id":"2402.07384","version":1},"attestation_state":"computed","paper":{"title":"Exploring Perceptual Limitation of Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Filip Ilievski, Jiarui Zhang, Jinyi Hu, Mahyar Khayatkhoei, Maosong Sun","submitted_at":"2024-02-12T03:04:42Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have recently shown remarkable perceptual capability in answering visual questions, however, little is known about the limits of their perception. In particular, while prior works have provided anecdotal evidence of MLLMs' sensitivity to object size, this phenomenon and its underlying causes have not been explored comprehensively. In this work, we quantitatively study the perception of small visual objects in several state-of-the-art MLLMs and reveal a pervasive limitation in answering questions about small objects in images. Next, we identify four inde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.07384","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-12T03:04:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6996924d2d1b11fdc2fa3aa09e7bb57886530c202e03f3cd94e71fbb3ec2b26b","abstract_canon_sha256":"03249e60eba5fe152a353e66b9e9f4e0ded986fc75e7f5bea018deaf62b54588"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:44:05.521128Z","signature_b64":"yQ0wL56Ce+5wbZwVTzToTlWddvt1i16olNiDo2l3/MNMwEyhUMqXO0TwuuNTTX27Md3x6fbjy07aUmnNUT/RDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cf886a740333569aa111cd590a01b2851c24e15d76644eb53d0ac6e9db460b7","last_reissued_at":"2026-07-05T07:44:05.520677Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:44:05.520677Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Perceptual Limitation of Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Filip Ilievski, Jiarui Zhang, Jinyi Hu, Mahyar Khayatkhoei, Maosong Sun","submitted_at":"2024-02-12T03:04:42Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have recently shown remarkable perceptual capability in answering visual questions, however, little is known about the limits of their perception. In particular, while prior works have provided anecdotal evidence of MLLMs' sensitivity to object size, this phenomenon and its underlying causes have not been explored comprehensively. In this work, we quantitatively study the perception of small visual objects in several state-of-the-art MLLMs and reveal a pervasive limitation in answering questions about small objects in images. Next, we identify four inde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07384","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.07384/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.07384","created_at":"2026-07-05T07:44:05.520729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.07384v1","created_at":"2026-07-05T07:44:05.520729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07384","created_at":"2026-07-05T07:44:05.520729+00:00"},{"alias_kind":"pith_short_12","alias_value":"FT4INJ2AGM2W","created_at":"2026-07-05T07:44:05.520729+00:00"},{"alias_kind":"pith_short_16","alias_value":"FT4INJ2AGM2WTKQR","created_at":"2026-07-05T07:44:05.520729+00:00"},{"alias_kind":"pith_short_8","alias_value":"FT4INJ2A","created_at":"2026-07-05T07:44:05.520729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21968","citing_title":"Look Before You Zoom: Adaptive Routing for the Resolution-Context Trade-off in Visual RAG","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20914","citing_title":"RISE: Reliable Improvement in Self-Evolving Vision-Language Models","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20914","citing_title":"RISE: Reliable Improvement in Self-Evolving Vision-Language Models","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB","json":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB.json","graph_json":"https://pith.science/api/pith-number/FT4INJ2AGM2WTKQRDTKZBIA3FB/graph.json","events_json":"https://pith.science/api/pith-number/FT4INJ2AGM2WTKQRDTKZBIA3FB/events.json","paper":"https://pith.science/paper/FT4INJ2A"},"agent_actions":{"view_html":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB","download_json":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB.json","view_paper":"https://pith.science/paper/FT4INJ2A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.07384&json=true","fetch_graph":"https://pith.science/api/pith-number/FT4INJ2AGM2WTKQRDTKZBIA3FB/graph.json","fetch_events":"https://pith.science/api/pith-number/FT4INJ2AGM2WTKQRDTKZBIA3FB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB/action/storage_attestation","attest_author":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB/action/author_attestation","sign_citation":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB/action/citation_signature","submit_replication":"https://pith.science/pith/FT4INJ2AGM2WTKQRDTKZBIA3FB/action/replication_record"}},"created_at":"2026-07-05T07:44:05.520729+00:00","updated_at":"2026-07-05T07:44:05.520729+00:00"}