{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:A7LFETKPKBZSGTEOPRSDSJ6FTR","short_pith_number":"pith:A7LFETKP","schema_version":"1.0","canonical_sha256":"07d6524d4f5073234c8e7c643927c59c773ba51f6834d3461964364b818bc019","source":{"kind":"arxiv","id":"2312.02896","version":2},"attestation_state":"computed","paper":{"title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Kot, Chenyu Yi, Dayan Guan, Rizhao Cai, Xing Luo, Zhenhao Chen, Zirui Song","submitted_at":"2023-12-05T17:06:59Z","abstract_excerpt":"Large Multimodal Models (LMMs) such as GPT-4V and LLaVA have shown remarkable capabilities in visual reasoning with common image styles. However, their robustness against diverse style shifts, crucial for practical applications, remains largely unexplored. In this paper, we propose a new benchmark, BenchLMM, to assess the robustness of LMMs against three different styles: artistic image style, imaging sensor style, and application style, where each style has five sub-styles. Utilizing BenchLMM, we comprehensively evaluate state-of-the-art LMMs and reveal: 1) LMMs generally suffer performance d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02896","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-05T17:06:59Z","cross_cats_sorted":[],"title_canon_sha256":"1f1c45df998c76b87f804c14cd483037e8b5a87abcff02796de2ce01c5acb9f5","abstract_canon_sha256":"f035fb69cd142395c0f0e970f16548fb73d051f9d1d7578b31ce83796de626c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:20:56.349747Z","signature_b64":"W88O8xE+W2XdjH2AOL7m0G8ms6H7hBNXtEdTg6AJtxF0bf2DG9FKAwc92jGSVLdM5z2bVmOz08Vc2HfNcnOWAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07d6524d4f5073234c8e7c643927c59c773ba51f6834d3461964364b818bc019","last_reissued_at":"2026-07-05T07:20:56.349219Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:20:56.349219Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Kot, Chenyu Yi, Dayan Guan, Rizhao Cai, Xing Luo, Zhenhao Chen, Zirui Song","submitted_at":"2023-12-05T17:06:59Z","abstract_excerpt":"Large Multimodal Models (LMMs) such as GPT-4V and LLaVA have shown remarkable capabilities in visual reasoning with common image styles. However, their robustness against diverse style shifts, crucial for practical applications, remains largely unexplored. In this paper, we propose a new benchmark, BenchLMM, to assess the robustness of LMMs against three different styles: artistic image style, imaging sensor style, and application style, where each style has five sub-styles. Utilizing BenchLMM, we comprehensively evaluate state-of-the-art LMMs and reveal: 1) LMMs generally suffer performance d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02896","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02896/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02896","created_at":"2026-07-05T07:20:56.349286+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02896v2","created_at":"2026-07-05T07:20:56.349286+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02896","created_at":"2026-07-05T07:20:56.349286+00:00"},{"alias_kind":"pith_short_12","alias_value":"A7LFETKPKBZS","created_at":"2026-07-05T07:20:56.349286+00:00"},{"alias_kind":"pith_short_16","alias_value":"A7LFETKPKBZSGTEO","created_at":"2026-07-05T07:20:56.349286+00:00"},{"alias_kind":"pith_short_8","alias_value":"A7LFETKP","created_at":"2026-07-05T07:20:56.349286+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR","json":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR.json","graph_json":"https://pith.science/api/pith-number/A7LFETKPKBZSGTEOPRSDSJ6FTR/graph.json","events_json":"https://pith.science/api/pith-number/A7LFETKPKBZSGTEOPRSDSJ6FTR/events.json","paper":"https://pith.science/paper/A7LFETKP"},"agent_actions":{"view_html":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR","download_json":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR.json","view_paper":"https://pith.science/paper/A7LFETKP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02896&json=true","fetch_graph":"https://pith.science/api/pith-number/A7LFETKPKBZSGTEOPRSDSJ6FTR/graph.json","fetch_events":"https://pith.science/api/pith-number/A7LFETKPKBZSGTEOPRSDSJ6FTR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR/action/storage_attestation","attest_author":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR/action/author_attestation","sign_citation":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR/action/citation_signature","submit_replication":"https://pith.science/pith/A7LFETKPKBZSGTEOPRSDSJ6FTR/action/replication_record"}},"created_at":"2026-07-05T07:20:56.349286+00:00","updated_at":"2026-07-05T07:20:56.349286+00:00"}