{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AZKA4JNVXAA7EMELDW726HPSKH","short_pith_number":"pith:AZKA4JNV","schema_version":"1.0","canonical_sha256":"06540e25b5b801f2308b1dbfaf1df251fe69ce38fb6d956b8dc67b825bc25d22","source":{"kind":"arxiv","id":"2412.00947","version":3},"attestation_state":"computed","paper":{"title":"VisOnlyQA: Large Vision Language Models Still Struggle with Visual Perception of Geometric Information","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Ranran Haoran Zhang, Rui Zhang, Ryo Kamoi, Sarkar Snigdha Sarathi Das, Yusen Zhang","submitted_at":"2024-12-01T19:46:22Z","abstract_excerpt":"Large Vision Language Models (LVLMs) have achieved remarkable performance in various vision-language tasks. However, it is still unclear how accurately LVLMs can perceive visual information in images. In particular, the capability of LVLMs to perceive geometric information, such as shape, angle, and size, remains insufficiently analyzed, although the perception of these properties is crucial for tasks that require a detailed visual understanding. In this work, we introduce VisOnlyQA, a dataset for evaluating the geometric perception of LVLMs, and reveal that LVLMs often cannot accurately perce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.00947","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-01T19:46:22Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"5f368cf116db4ca81ff8411bde139898167d4757ce407681dee01864055d6cb8","abstract_canon_sha256":"2e0fbe33d1516696d5243df6677de7868a40b9dbad0f44cc6b90ac5e4fd387b9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:25.979022Z","signature_b64":"9A4LnhzjMsc1z4X4IRhUIkXBWtaCLMsr3bFVyiDsq3vj1PRMihZ0Ym+31UdVj4u1bjSp6qy10zUjJETyWPDiAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06540e25b5b801f2308b1dbfaf1df251fe69ce38fb6d956b8dc67b825bc25d22","last_reissued_at":"2026-07-05T11:36:25.978529Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:25.978529Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisOnlyQA: Large Vision Language Models Still Struggle with Visual Perception of Geometric Information","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Ranran Haoran Zhang, Rui Zhang, Ryo Kamoi, Sarkar Snigdha Sarathi Das, Yusen Zhang","submitted_at":"2024-12-01T19:46:22Z","abstract_excerpt":"Large Vision Language Models (LVLMs) have achieved remarkable performance in various vision-language tasks. However, it is still unclear how accurately LVLMs can perceive visual information in images. In particular, the capability of LVLMs to perceive geometric information, such as shape, angle, and size, remains insufficiently analyzed, although the perception of these properties is crucial for tasks that require a detailed visual understanding. In this work, we introduce VisOnlyQA, a dataset for evaluating the geometric perception of LVLMs, and reveal that LVLMs often cannot accurately perce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.00947","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.00947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.00947","created_at":"2026-07-05T11:36:25.978594+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.00947v3","created_at":"2026-07-05T11:36:25.978594+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.00947","created_at":"2026-07-05T11:36:25.978594+00:00"},{"alias_kind":"pith_short_12","alias_value":"AZKA4JNVXAA7","created_at":"2026-07-05T11:36:25.978594+00:00"},{"alias_kind":"pith_short_16","alias_value":"AZKA4JNVXAA7EMEL","created_at":"2026-07-05T11:36:25.978594+00:00"},{"alias_kind":"pith_short_8","alias_value":"AZKA4JNV","created_at":"2026-07-05T11:36:25.978594+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07653","citing_title":"A Dataset for Dynamic Human Preferences for Vision Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20177","citing_title":"From Seeing to Thinking: Decoupling Perception and Reasoning Improves Post-Training of Vision-Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26614","citing_title":"State Beyond Appearance: Diagnosing and Improving State Consistency in Dial-Based Measurement Reading","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13321","citing_title":"Why MLLMs Struggle to Determine Object Orientations","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH","json":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH.json","graph_json":"https://pith.science/api/pith-number/AZKA4JNVXAA7EMELDW726HPSKH/graph.json","events_json":"https://pith.science/api/pith-number/AZKA4JNVXAA7EMELDW726HPSKH/events.json","paper":"https://pith.science/paper/AZKA4JNV"},"agent_actions":{"view_html":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH","download_json":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH.json","view_paper":"https://pith.science/paper/AZKA4JNV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.00947&json=true","fetch_graph":"https://pith.science/api/pith-number/AZKA4JNVXAA7EMELDW726HPSKH/graph.json","fetch_events":"https://pith.science/api/pith-number/AZKA4JNVXAA7EMELDW726HPSKH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH/action/storage_attestation","attest_author":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH/action/author_attestation","sign_citation":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH/action/citation_signature","submit_replication":"https://pith.science/pith/AZKA4JNVXAA7EMELDW726HPSKH/action/replication_record"}},"created_at":"2026-07-05T11:36:25.978594+00:00","updated_at":"2026-07-05T11:36:25.978594+00:00"}