{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TUKJOST4H3OS74YHWKWHHTCY5V","short_pith_number":"pith:TUKJOST4","schema_version":"1.0","canonical_sha256":"9d14974a7c3edd2ff307b2ac73cc58ed5c3171f8f3603f3f39c06c3bde2512c9","source":{"kind":"arxiv","id":"2412.08169","version":1},"attestation_state":"computed","paper":{"title":"Illusory VQA: Benchmarking and Enhancing Multimodal Models on Visual Illusions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Baktash Ansari, Farzan Rahmani, Hoorieh Sabzevari, Mohammadmostafa Rostamkhani, Sauleh Eetemadi","submitted_at":"2024-12-11T07:51:18Z","abstract_excerpt":"In recent years, Visual Question Answering (VQA) has made significant strides, particularly with the advent of multimodal models that integrate vision and language understanding. However, existing VQA datasets often overlook the complexities introduced by image illusions, which pose unique challenges for both human perception and model interpretation. In this study, we introduce a novel task called Illusory VQA, along with four specialized datasets: IllusionMNIST, IllusionFashionMNIST, IllusionAnimals, and IllusionChar. These datasets are designed to evaluate the performance of state-of-the-ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08169","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-11T07:51:18Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"135c70ee8858149294ea49ff5ac3d3f075bc58d5dbfb599a579b73d45d76f0f7","abstract_canon_sha256":"c8b8812a91e6fa5b0e78b13326937485fc4999aae73e42c1228565333c837eca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:45.506439Z","signature_b64":"+MyCpL2qNYQTtFJh/rSBK4TnzQ0yhToqjSkH7ieK90IKNZjUv91DpWxZ90UtRIp08sIGaiidb3jn9/cF7xPSCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d14974a7c3edd2ff307b2ac73cc58ed5c3171f8f3603f3f39c06c3bde2512c9","last_reissued_at":"2026-07-05T09:47:45.505949Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:45.505949Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Illusory VQA: Benchmarking and Enhancing Multimodal Models on Visual Illusions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Baktash Ansari, Farzan Rahmani, Hoorieh Sabzevari, Mohammadmostafa Rostamkhani, Sauleh Eetemadi","submitted_at":"2024-12-11T07:51:18Z","abstract_excerpt":"In recent years, Visual Question Answering (VQA) has made significant strides, particularly with the advent of multimodal models that integrate vision and language understanding. However, existing VQA datasets often overlook the complexities introduced by image illusions, which pose unique challenges for both human perception and model interpretation. In this study, we introduce a novel task called Illusory VQA, along with four specialized datasets: IllusionMNIST, IllusionFashionMNIST, IllusionAnimals, and IllusionChar. These datasets are designed to evaluate the performance of state-of-the-ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08169","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08169/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08169","created_at":"2026-07-05T09:47:45.506010+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08169v1","created_at":"2026-07-05T09:47:45.506010+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08169","created_at":"2026-07-05T09:47:45.506010+00:00"},{"alias_kind":"pith_short_12","alias_value":"TUKJOST4H3OS","created_at":"2026-07-05T09:47:45.506010+00:00"},{"alias_kind":"pith_short_16","alias_value":"TUKJOST4H3OS74YH","created_at":"2026-07-05T09:47:45.506010+00:00"},{"alias_kind":"pith_short_8","alias_value":"TUKJOST4","created_at":"2026-07-05T09:47:45.506010+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.22617","citing_title":"Hate in Plain Sight: On the Risks of Moderating AI-Generated Hateful Illusions","ref_index":55,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V","json":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V.json","graph_json":"https://pith.science/api/pith-number/TUKJOST4H3OS74YHWKWHHTCY5V/graph.json","events_json":"https://pith.science/api/pith-number/TUKJOST4H3OS74YHWKWHHTCY5V/events.json","paper":"https://pith.science/paper/TUKJOST4"},"agent_actions":{"view_html":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V","download_json":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V.json","view_paper":"https://pith.science/paper/TUKJOST4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08169&json=true","fetch_graph":"https://pith.science/api/pith-number/TUKJOST4H3OS74YHWKWHHTCY5V/graph.json","fetch_events":"https://pith.science/api/pith-number/TUKJOST4H3OS74YHWKWHHTCY5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V/action/storage_attestation","attest_author":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V/action/author_attestation","sign_citation":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V/action/citation_signature","submit_replication":"https://pith.science/pith/TUKJOST4H3OS74YHWKWHHTCY5V/action/replication_record"}},"created_at":"2026-07-05T09:47:45.506010+00:00","updated_at":"2026-07-05T09:47:45.506010+00:00"}