{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OXVMEL4XAEHXYVL5O5MUZUEYND","short_pith_number":"pith:OXVMEL4X","schema_version":"1.0","canonical_sha256":"75eac22f97010f7c557d77594cd09868fdb8c9801b392d0278d9ba327d93c5bb","source":{"kind":"arxiv","id":"2308.03729","version":2},"attestation_state":"computed","paper":{"title":"TinyLVLM-eHub: Towards Comprehensive and Efficient Evaluation for Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fanqing Meng, Hongsheng Li, Kaipeng Zhang, Meng Lei, Peng Gao, Peng Xu, Ping Luo, Siyuan Huang, Wenqi Shao, Yu Qiao, Yutao Hu","submitted_at":"2023-08-07T17:17:05Z","abstract_excerpt":"Recent advancements in Large Vision-Language Models (LVLMs) have demonstrated significant progress in tackling complex multimodal tasks. Among these cutting-edge developments, Google's Bard stands out for its remarkable multimodal capabilities, promoting comprehensive comprehension and reasoning across various domains. This work presents an early and holistic evaluation of LVLMs' multimodal abilities, with a particular focus on Bard, by proposing a lightweight variant of LVLM-eHub, named Tiny LVLM-eHub. In comparison to the vanilla version, Tiny LVLM-eHub possesses several appealing properties"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.03729","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-07T17:17:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"009fe4cd9910ff275bf089117d3f39a560fbbcc839c830a9e427a53716a33f97","abstract_canon_sha256":"27cd1b3a9495189478fbfadf1ebc89bc6d30b1e4013bda72973f810f0d37708a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:03.047916Z","signature_b64":"din/gUE4O3+FYfXqmSb8S7arCg4XLmmOhtIgcwTIVcBKqZ4WANv7wJOc/7UX5VD1XFs4xNY6P5s1YaDIF3I7Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75eac22f97010f7c557d77594cd09868fdb8c9801b392d0278d9ba327d93c5bb","last_reissued_at":"2026-07-05T08:54:03.047551Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:03.047551Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TinyLVLM-eHub: Towards Comprehensive and Efficient Evaluation for Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fanqing Meng, Hongsheng Li, Kaipeng Zhang, Meng Lei, Peng Gao, Peng Xu, Ping Luo, Siyuan Huang, Wenqi Shao, Yu Qiao, Yutao Hu","submitted_at":"2023-08-07T17:17:05Z","abstract_excerpt":"Recent advancements in Large Vision-Language Models (LVLMs) have demonstrated significant progress in tackling complex multimodal tasks. Among these cutting-edge developments, Google's Bard stands out for its remarkable multimodal capabilities, promoting comprehensive comprehension and reasoning across various domains. This work presents an early and holistic evaluation of LVLMs' multimodal abilities, with a particular focus on Bard, by proposing a lightweight variant of LVLM-eHub, named Tiny LVLM-eHub. In comparison to the vanilla version, Tiny LVLM-eHub possesses several appealing properties"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.03729","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.03729/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.03729","created_at":"2026-07-05T08:54:03.047606+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.03729v2","created_at":"2026-07-05T08:54:03.047606+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.03729","created_at":"2026-07-05T08:54:03.047606+00:00"},{"alias_kind":"pith_short_12","alias_value":"OXVMEL4XAEHX","created_at":"2026-07-05T08:54:03.047606+00:00"},{"alias_kind":"pith_short_16","alias_value":"OXVMEL4XAEHXYVL5","created_at":"2026-07-05T08:54:03.047606+00:00"},{"alias_kind":"pith_short_8","alias_value":"OXVMEL4X","created_at":"2026-07-05T08:54:03.047606+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.14702","citing_title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2309.15112","citing_title":"InternLM-XComposer: A Vision-Language Large Model for Advanced Text-image Comprehension and Composition","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07575","citing_title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":124,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND","json":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND.json","graph_json":"https://pith.science/api/pith-number/OXVMEL4XAEHXYVL5O5MUZUEYND/graph.json","events_json":"https://pith.science/api/pith-number/OXVMEL4XAEHXYVL5O5MUZUEYND/events.json","paper":"https://pith.science/paper/OXVMEL4X"},"agent_actions":{"view_html":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND","download_json":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND.json","view_paper":"https://pith.science/paper/OXVMEL4X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.03729&json=true","fetch_graph":"https://pith.science/api/pith-number/OXVMEL4XAEHXYVL5O5MUZUEYND/graph.json","fetch_events":"https://pith.science/api/pith-number/OXVMEL4XAEHXYVL5O5MUZUEYND/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND/action/storage_attestation","attest_author":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND/action/author_attestation","sign_citation":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND/action/citation_signature","submit_replication":"https://pith.science/pith/OXVMEL4XAEHXYVL5O5MUZUEYND/action/replication_record"}},"created_at":"2026-07-05T08:54:03.047606+00:00","updated_at":"2026-07-05T08:54:03.047606+00:00"}