{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZOPNPSNWZ2U3J2U3V7PVNFUI3P","short_pith_number":"pith:ZOPNPSNW","schema_version":"1.0","canonical_sha256":"cb9ed7c9b6cea9b4ea9bafdf569688dbf2e177baefcf41e44d5851f3c26e8f5a","source":{"kind":"arxiv","id":"2410.00193","version":3},"attestation_state":"computed","paper":{"title":"Do Vision-Language Models Really Understand Visual Language?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Buse Giledereli, Mrinmaya Sachan, Yifan Hou, Yilei Tu","submitted_at":"2024-09-30T19:45:11Z","abstract_excerpt":"Visual language is a system of communication that conveys information through symbols, shapes, and spatial arrangements. Diagrams are a typical example of a visual language depicting complex concepts and their relationships in the form of an image. The symbolic nature of diagrams presents significant challenges for building models capable of understanding them. Recent studies suggest that Large Vision-Language Models (LVLMs) can even tackle complex reasoning tasks involving diagrams. In this paper, we investigate this phenomenon by developing a comprehensive test suite to evaluate the diagram "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.00193","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-30T19:45:11Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"16c775d3f42044c8f1cba72aede6b1511805e676d5183a8c155aaa496e142a1d","abstract_canon_sha256":"e8ee8e03bc3a847869f155b0022bcc288d082e01fa7e21d3a030e372707ed703"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:00.908306Z","signature_b64":"9IMPr7i7rYRXFFe8uoxwKvUA3E8GxEtG5LAF+SiuSXdJEkDkxsuPYnbAhVvesoUNfncB2x/K1qc6a+8NGv8TAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb9ed7c9b6cea9b4ea9bafdf569688dbf2e177baefcf41e44d5851f3c26e8f5a","last_reissued_at":"2026-07-05T11:09:00.907857Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:00.907857Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Vision-Language Models Really Understand Visual Language?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Buse Giledereli, Mrinmaya Sachan, Yifan Hou, Yilei Tu","submitted_at":"2024-09-30T19:45:11Z","abstract_excerpt":"Visual language is a system of communication that conveys information through symbols, shapes, and spatial arrangements. Diagrams are a typical example of a visual language depicting complex concepts and their relationships in the form of an image. The symbolic nature of diagrams presents significant challenges for building models capable of understanding them. Recent studies suggest that Large Vision-Language Models (LVLMs) can even tackle complex reasoning tasks involving diagrams. In this paper, we investigate this phenomenon by developing a comprehensive test suite to evaluate the diagram "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.00193","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.00193/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.00193","created_at":"2026-07-05T11:09:00.907913+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.00193v3","created_at":"2026-07-05T11:09:00.907913+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.00193","created_at":"2026-07-05T11:09:00.907913+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZOPNPSNWZ2U3","created_at":"2026-07-05T11:09:00.907913+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZOPNPSNWZ2U3J2U3","created_at":"2026-07-05T11:09:00.907913+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZOPNPSNW","created_at":"2026-07-05T11:09:00.907913+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20177","citing_title":"From Seeing to Thinking: Decoupling Perception and Reasoning Improves Post-Training of Vision-Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17722","citing_title":"Can Vision-Language Models Count? A Synthetic Benchmark and Analysis of Attention-Based Interventions","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04009","citing_title":"Benchmarking and Evaluating VLMs for Software Architecture Diagram Understanding","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P","json":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P.json","graph_json":"https://pith.science/api/pith-number/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/graph.json","events_json":"https://pith.science/api/pith-number/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/events.json","paper":"https://pith.science/paper/ZOPNPSNW"},"agent_actions":{"view_html":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P","download_json":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P.json","view_paper":"https://pith.science/paper/ZOPNPSNW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.00193&json=true","fetch_graph":"https://pith.science/api/pith-number/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/graph.json","fetch_events":"https://pith.science/api/pith-number/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/action/storage_attestation","attest_author":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/action/author_attestation","sign_citation":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/action/citation_signature","submit_replication":"https://pith.science/pith/ZOPNPSNWZ2U3J2U3V7PVNFUI3P/action/replication_record"}},"created_at":"2026-07-05T11:09:00.907913+00:00","updated_at":"2026-07-05T11:09:00.907913+00:00"}