{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AWJMYB4YQPJCC6WFTVWXRHJGS7","short_pith_number":"pith:AWJMYB4Y","schema_version":"1.0","canonical_sha256":"0592cc079883d2217ac59d6d789d2697ec35d2d36ea07610b76ee8d28147a360","source":{"kind":"arxiv","id":"2410.21259","version":4},"attestation_state":"computed","paper":{"title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Han Bao, Jiayi Ye, Mohamed Elhoseiny, Tianyi Zhou, Xiangliang Zhang, Xiangqi Wang, Xiuying Chen, Yanbo Wang, Yue Huang, Yue Zhao","submitted_at":"2024-10-28T17:55:08Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have become essential for advancing the integration of visual and linguistic information. However, the evaluation of LVLMs presents significant challenges as the evaluation benchmark always demands lots of human cost for its construction, and remains static, lacking flexibility once constructed. Even though automatic evaluation has been explored in textual modality, the visual modality remains under-explored. As a result, in this work, we address a question: \"Can LVLMs themselves be used to benchmark each other in the visual automatically domain?\". We intro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21259","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-28T17:55:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8d26e6884e8f2af2536a4ff784277a527fe104eb81fdfc9f9ab17ea78ec7c007","abstract_canon_sha256":"7d18652aa7d1c19998a32cb32f53851faa6fb72dcbf690d05e559c9db057054a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:12.980452Z","signature_b64":"mv3wdeYpBwwVBpfw/GxlI84v4gdK3Tx2YSrOecO/TA8stmWEQ3iI6ueO0yzL5fIHMVOy4KlJy1ur/jCcotM5BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0592cc079883d2217ac59d6d789d2697ec35d2d36ea07610b76ee8d28147a360","last_reissued_at":"2026-07-05T10:25:12.979962Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:12.979962Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Han Bao, Jiayi Ye, Mohamed Elhoseiny, Tianyi Zhou, Xiangliang Zhang, Xiangqi Wang, Xiuying Chen, Yanbo Wang, Yue Huang, Yue Zhao","submitted_at":"2024-10-28T17:55:08Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have become essential for advancing the integration of visual and linguistic information. However, the evaluation of LVLMs presents significant challenges as the evaluation benchmark always demands lots of human cost for its construction, and remains static, lacking flexibility once constructed. Even though automatic evaluation has been explored in textual modality, the visual modality remains under-explored. As a result, in this work, we address a question: \"Can LVLMs themselves be used to benchmark each other in the visual automatically domain?\". We intro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21259","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21259/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21259","created_at":"2026-07-05T10:25:12.980019+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21259v4","created_at":"2026-07-05T10:25:12.980019+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21259","created_at":"2026-07-05T10:25:12.980019+00:00"},{"alias_kind":"pith_short_12","alias_value":"AWJMYB4YQPJC","created_at":"2026-07-05T10:25:12.980019+00:00"},{"alias_kind":"pith_short_16","alias_value":"AWJMYB4YQPJCC6WF","created_at":"2026-07-05T10:25:12.980019+00:00"},{"alias_kind":"pith_short_8","alias_value":"AWJMYB4Y","created_at":"2026-07-05T10:25:12.980019+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20736","citing_title":"REKEY: Metadata-Grounded Visual-Key Regeneration for Contamination-Resilient VQA Evaluation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10999","citing_title":"SkillGen: Verified Inference-Time Agent Skill Synthesis","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12995","citing_title":"PolicyLLM: Towards Excellent Comprehension of Public Policy for Large Language Models","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7","json":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7.json","graph_json":"https://pith.science/api/pith-number/AWJMYB4YQPJCC6WFTVWXRHJGS7/graph.json","events_json":"https://pith.science/api/pith-number/AWJMYB4YQPJCC6WFTVWXRHJGS7/events.json","paper":"https://pith.science/paper/AWJMYB4Y"},"agent_actions":{"view_html":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7","download_json":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7.json","view_paper":"https://pith.science/paper/AWJMYB4Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21259&json=true","fetch_graph":"https://pith.science/api/pith-number/AWJMYB4YQPJCC6WFTVWXRHJGS7/graph.json","fetch_events":"https://pith.science/api/pith-number/AWJMYB4YQPJCC6WFTVWXRHJGS7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7/action/storage_attestation","attest_author":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7/action/author_attestation","sign_citation":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7/action/citation_signature","submit_replication":"https://pith.science/pith/AWJMYB4YQPJCC6WFTVWXRHJGS7/action/replication_record"}},"created_at":"2026-07-05T10:25:12.980019+00:00","updated_at":"2026-07-05T10:25:12.980019+00:00"}