{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:COOK5QG24E5VGLXUDHD6JBHMVG","short_pith_number":"pith:COOK5QG2","schema_version":"1.0","canonical_sha256":"139caec0dae13b532ef419c7e484eca9a2d6fb9a3f1891d9ea12c554a73cb991","source":{"kind":"arxiv","id":"2410.12381","version":3},"attestation_state":"computed","paper":{"title":"HumanEval-V: Benchmarking High-Level Visual Reasoning with Complex Diagrams in Coding Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bei Chen, Fengji Zhang, Guancheng Lin, Huiyu Bai, Jacky Keung, Linquan Wu, Xiao Li, Xiao Yu, Yue Wang","submitted_at":"2024-10-16T09:04:57Z","abstract_excerpt":"Understanding and reasoning over diagrams is a fundamental aspect of human intelligence. While Large Multimodal Models (LMMs) have demonstrated impressive capabilities across various tasks, existing benchmarks lack comprehensive evaluation of their diagram interpretation and reasoning abilities, particularly in coding contexts. We present HumanEval-V, a rigorous benchmark of human-annotated coding tasks that spans six task types and evaluates diverse visual reasoning capabilities. Each task features carefully crafted diagrams paired with function signatures and test cases, employing novel code"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12381","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-16T09:04:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2119ba52e7e1ffbf4c5312c22f82a4bf9e4106be6504a58fdf7e1e2ce7c85e79","abstract_canon_sha256":"c49db77b7946f99eb6d925884fe3f2ed816c513f1f696dabf953f15e9eb02a4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:46.356036Z","signature_b64":"fCMP9UbfTgBhquJO4ienQHZi5WyVjb3cwFbbyq4MulS8IXNdQkUTyF3NiZHy7dYbcBTb7CkN7181spsm8JeTAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"139caec0dae13b532ef419c7e484eca9a2d6fb9a3f1891d9ea12c554a73cb991","last_reissued_at":"2026-07-05T10:15:46.355456Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:46.355456Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HumanEval-V: Benchmarking High-Level Visual Reasoning with Complex Diagrams in Coding Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bei Chen, Fengji Zhang, Guancheng Lin, Huiyu Bai, Jacky Keung, Linquan Wu, Xiao Li, Xiao Yu, Yue Wang","submitted_at":"2024-10-16T09:04:57Z","abstract_excerpt":"Understanding and reasoning over diagrams is a fundamental aspect of human intelligence. While Large Multimodal Models (LMMs) have demonstrated impressive capabilities across various tasks, existing benchmarks lack comprehensive evaluation of their diagram interpretation and reasoning abilities, particularly in coding contexts. We present HumanEval-V, a rigorous benchmark of human-annotated coding tasks that spans six task types and evaluates diverse visual reasoning capabilities. Each task features carefully crafted diagrams paired with function signatures and test cases, employing novel code"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12381","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12381/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12381","created_at":"2026-07-05T10:15:46.355544+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12381v3","created_at":"2026-07-05T10:15:46.355544+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12381","created_at":"2026-07-05T10:15:46.355544+00:00"},{"alias_kind":"pith_short_12","alias_value":"COOK5QG24E5V","created_at":"2026-07-05T10:15:46.355544+00:00"},{"alias_kind":"pith_short_16","alias_value":"COOK5QG24E5VGLXU","created_at":"2026-07-05T10:15:46.355544+00:00"},{"alias_kind":"pith_short_8","alias_value":"COOK5QG2","created_at":"2026-07-05T10:15:46.355544+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05769","citing_title":"Imagine Before You Predict: Interleaved Latent Visual Reasoning for Video Event Prediction","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22177","citing_title":"Maestro: Reinforcement Learning to Orchestrate Hierarchical Model-Skill Ensembles","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2503.10615","citing_title":"R1-Onevision: Advancing Generalized Multimodal Reasoning through Cross-Modal Formalization","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG","json":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG.json","graph_json":"https://pith.science/api/pith-number/COOK5QG24E5VGLXUDHD6JBHMVG/graph.json","events_json":"https://pith.science/api/pith-number/COOK5QG24E5VGLXUDHD6JBHMVG/events.json","paper":"https://pith.science/paper/COOK5QG2"},"agent_actions":{"view_html":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG","download_json":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG.json","view_paper":"https://pith.science/paper/COOK5QG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12381&json=true","fetch_graph":"https://pith.science/api/pith-number/COOK5QG24E5VGLXUDHD6JBHMVG/graph.json","fetch_events":"https://pith.science/api/pith-number/COOK5QG24E5VGLXUDHD6JBHMVG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG/action/storage_attestation","attest_author":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG/action/author_attestation","sign_citation":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG/action/citation_signature","submit_replication":"https://pith.science/pith/COOK5QG24E5VGLXUDHD6JBHMVG/action/replication_record"}},"created_at":"2026-07-05T10:15:46.355544+00:00","updated_at":"2026-07-05T10:15:46.355544+00:00"}