{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BFYZUWUXETGIII24TQKJ652EBW","short_pith_number":"pith:BFYZUWUX","schema_version":"1.0","canonical_sha256":"09719a5a9724cc84235c9c149f77440d98ece3bf00a36e5daa228bde207ec60d","source":{"kind":"arxiv","id":"2412.16418","version":1},"attestation_state":"computed","paper":{"title":"Revisiting MLLMs: An In-Depth Analysis of Image Classification Abilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huan Liu, Jiangjiang Liu, Jingdong Wang, Lingyu Xiao, Sen Yang, Xiaofan Li, Ze Feng","submitted_at":"2024-12-21T00:46:56Z","abstract_excerpt":"With the rapid advancement of Multimodal Large Language Models (MLLMs), a variety of benchmarks have been introduced to evaluate their capabilities. While most evaluations have focused on complex tasks such as scientific comprehension and visual reasoning, little attention has been given to assessing their fundamental image classification abilities. In this paper, we address this gap by thoroughly revisiting the MLLMs with an in-depth analysis of image classification. Specifically, building on established datasets, we examine a broad spectrum of scenarios, from general classification tasks (e."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16418","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-21T00:46:56Z","cross_cats_sorted":[],"title_canon_sha256":"1d860bed6887b24eaa11149d31839154016e2f19eb195b97ce4de6ff61dd27d4","abstract_canon_sha256":"0a0971b8f6b666482cc758e5131ca71bfd47b00106cc710a241dbab7edb33388"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:45.732977Z","signature_b64":"h0Jwn3vGkWL+aRu6q+wHjx2Nli9oFCk8EfQCr6yH/AJAlpOQjACyU0zIj8wFjN7IK3xhdOyfPM1RqmPmCIKODA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09719a5a9724cc84235c9c149f77440d98ece3bf00a36e5daa228bde207ec60d","last_reissued_at":"2026-07-05T09:52:45.732525Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:45.732525Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting MLLMs: An In-Depth Analysis of Image Classification Abilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huan Liu, Jiangjiang Liu, Jingdong Wang, Lingyu Xiao, Sen Yang, Xiaofan Li, Ze Feng","submitted_at":"2024-12-21T00:46:56Z","abstract_excerpt":"With the rapid advancement of Multimodal Large Language Models (MLLMs), a variety of benchmarks have been introduced to evaluate their capabilities. While most evaluations have focused on complex tasks such as scientific comprehension and visual reasoning, little attention has been given to assessing their fundamental image classification abilities. In this paper, we address this gap by thoroughly revisiting the MLLMs with an in-depth analysis of image classification. Specifically, building on established datasets, we examine a broad spectrum of scenarios, from general classification tasks (e."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16418","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16418","created_at":"2026-07-05T09:52:45.732587+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16418v1","created_at":"2026-07-05T09:52:45.732587+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16418","created_at":"2026-07-05T09:52:45.732587+00:00"},{"alias_kind":"pith_short_12","alias_value":"BFYZUWUXETGI","created_at":"2026-07-05T09:52:45.732587+00:00"},{"alias_kind":"pith_short_16","alias_value":"BFYZUWUXETGIII24","created_at":"2026-07-05T09:52:45.732587+00:00"},{"alias_kind":"pith_short_8","alias_value":"BFYZUWUX","created_at":"2026-07-05T09:52:45.732587+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.04326","citing_title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07605","citing_title":"Fine-R1: Make Multi-modal LLMs Excel in Fine-Grained Visual Recognition by Chain-of-Thought Reasoning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03197","citing_title":"Specificity-aware reinforcement learning for fine-grained open-world classification","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16785","citing_title":"Bridging Coarse and Fine Recognition: A Hybrid Approach for Open-Ended Multi-Granularity Object Recognition in Interactive Educational Games","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW","json":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW.json","graph_json":"https://pith.science/api/pith-number/BFYZUWUXETGIII24TQKJ652EBW/graph.json","events_json":"https://pith.science/api/pith-number/BFYZUWUXETGIII24TQKJ652EBW/events.json","paper":"https://pith.science/paper/BFYZUWUX"},"agent_actions":{"view_html":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW","download_json":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW.json","view_paper":"https://pith.science/paper/BFYZUWUX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16418&json=true","fetch_graph":"https://pith.science/api/pith-number/BFYZUWUXETGIII24TQKJ652EBW/graph.json","fetch_events":"https://pith.science/api/pith-number/BFYZUWUXETGIII24TQKJ652EBW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW/action/storage_attestation","attest_author":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW/action/author_attestation","sign_citation":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW/action/citation_signature","submit_replication":"https://pith.science/pith/BFYZUWUXETGIII24TQKJ652EBW/action/replication_record"}},"created_at":"2026-07-05T09:52:45.732587+00:00","updated_at":"2026-07-05T09:52:45.732587+00:00"}