{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FWEICE2OPY7DCYU4ABLNLT7X4X","short_pith_number":"pith:FWEICE2O","schema_version":"1.0","canonical_sha256":"2d8881134e7e3e31629c0056d5cff7e5cc981a8879263825d4c6aebed1ba2b75","source":{"kind":"arxiv","id":"2410.11437","version":1},"attestation_state":"computed","paper":{"title":"Difficult Task Yes but Simple Task No: Unveiling the Laziness in Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Pinjia He, Sihang Zhao, Xiaoying Tang, Youliang Yuan","submitted_at":"2024-10-15T09:40:50Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) demonstrate a strong understanding of the real world and can even handle complex tasks. However, they still fail on some straightforward visual question-answering (VQA) problems. This paper dives deeper into this issue, revealing that models tend to err when answering easy questions (e.g. Yes/No questions) about an image, even though they can correctly describe it. We refer to this model behavior discrepancy between difficult and simple questions as model laziness. To systematically investigate model laziness, we manually construct LazyBench, a benchmar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11437","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-15T09:40:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"61a2faa4b62134365b48f4ba535d175468c6e94bf5009ffa0ce2dcc1e190f3a9","abstract_canon_sha256":"8a9375006947cb6b13de7b8008d9b99a1cc0d959a9e6e8bb2893f794929b62c0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:53.667415Z","signature_b64":"Ea+ibRu29L+1n5Nsh9XIxifaOECjpVK2QPh9HMgC6FwyTdHFR2Bo2BNPUjONF8GZ6U5UrqREZLMOlfdtiv8zDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d8881134e7e3e31629c0056d5cff7e5cc981a8879263825d4c6aebed1ba2b75","last_reissued_at":"2026-07-05T09:20:53.666956Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:53.666956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Difficult Task Yes but Simple Task No: Unveiling the Laziness in Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Pinjia He, Sihang Zhao, Xiaoying Tang, Youliang Yuan","submitted_at":"2024-10-15T09:40:50Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) demonstrate a strong understanding of the real world and can even handle complex tasks. However, they still fail on some straightforward visual question-answering (VQA) problems. This paper dives deeper into this issue, revealing that models tend to err when answering easy questions (e.g. Yes/No questions) about an image, even though they can correctly describe it. We refer to this model behavior discrepancy between difficult and simple questions as model laziness. To systematically investigate model laziness, we manually construct LazyBench, a benchmar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11437","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11437/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11437","created_at":"2026-07-05T09:20:53.667008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11437v1","created_at":"2026-07-05T09:20:53.667008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11437","created_at":"2026-07-05T09:20:53.667008+00:00"},{"alias_kind":"pith_short_12","alias_value":"FWEICE2OPY7D","created_at":"2026-07-05T09:20:53.667008+00:00"},{"alias_kind":"pith_short_16","alias_value":"FWEICE2OPY7DCYU4","created_at":"2026-07-05T09:20:53.667008+00:00"},{"alias_kind":"pith_short_8","alias_value":"FWEICE2O","created_at":"2026-07-05T09:20:53.667008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X","json":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X.json","graph_json":"https://pith.science/api/pith-number/FWEICE2OPY7DCYU4ABLNLT7X4X/graph.json","events_json":"https://pith.science/api/pith-number/FWEICE2OPY7DCYU4ABLNLT7X4X/events.json","paper":"https://pith.science/paper/FWEICE2O"},"agent_actions":{"view_html":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X","download_json":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X.json","view_paper":"https://pith.science/paper/FWEICE2O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11437&json=true","fetch_graph":"https://pith.science/api/pith-number/FWEICE2OPY7DCYU4ABLNLT7X4X/graph.json","fetch_events":"https://pith.science/api/pith-number/FWEICE2OPY7DCYU4ABLNLT7X4X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X/action/storage_attestation","attest_author":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X/action/author_attestation","sign_citation":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X/action/citation_signature","submit_replication":"https://pith.science/pith/FWEICE2OPY7DCYU4ABLNLT7X4X/action/replication_record"}},"created_at":"2026-07-05T09:20:53.667008+00:00","updated_at":"2026-07-05T09:20:53.667008+00:00"}