{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QRQR2TJZ5KJNORLW3OVGPX5OSH","short_pith_number":"pith:QRQR2TJZ","schema_version":"1.0","canonical_sha256":"84611d4d39ea92d74576dbaa67dfae91dbcac7cd3201724bc9ad243e85d10cee","source":{"kind":"arxiv","id":"2501.03939","version":2},"attestation_state":"computed","paper":{"title":"Visual question answering: from early developments to recent advances -- a survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Hakim Hacid, Imran Razzak, Mohamed Reda Bouadjenek, Ngoc Dung Huynh, Sunil Aryal","submitted_at":"2025-01-07T17:00:35Z","abstract_excerpt":"Visual Question Answering (VQA) is an evolving research field aimed at enabling machines to answer questions about visual content by integrating image and language processing techniques such as feature extraction, object detection, text embedding, natural language understanding, and language generation. With the growth of multimodal data research, VQA has gained significant attention due to its broad applications, including interactive educational tools, medical image diagnosis, customer service, entertainment, and social media captioning. Additionally, VQA plays a vital role in assisting visu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03939","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-07T17:00:35Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"4b3c70aa75bf9ee6d520f11fb480a7b188b8184e91e3164ff147f555893b52a2","abstract_canon_sha256":"0769df7aa810d0d41d5fdc8d23d8f84b1770d4a64b35b446e32f6d17e772b393"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:59.131946Z","signature_b64":"2gvVnhuMS4GYIj9lFLZv7gUkMdxJQ12zmT+gdRmx/fV6BAMJrwHMtDnY/Vrx6iyMOSePbLo28K6fnyJhkUk0BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84611d4d39ea92d74576dbaa67dfae91dbcac7cd3201724bc9ad243e85d10cee","last_reissued_at":"2026-07-05T09:59:59.131481Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:59.131481Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual question answering: from early developments to recent advances -- a survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Hakim Hacid, Imran Razzak, Mohamed Reda Bouadjenek, Ngoc Dung Huynh, Sunil Aryal","submitted_at":"2025-01-07T17:00:35Z","abstract_excerpt":"Visual Question Answering (VQA) is an evolving research field aimed at enabling machines to answer questions about visual content by integrating image and language processing techniques such as feature extraction, object detection, text embedding, natural language understanding, and language generation. With the growth of multimodal data research, VQA has gained significant attention due to its broad applications, including interactive educational tools, medical image diagnosis, customer service, entertainment, and social media captioning. Additionally, VQA plays a vital role in assisting visu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03939","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03939/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03939","created_at":"2026-07-05T09:59:59.131537+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03939v2","created_at":"2026-07-05T09:59:59.131537+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03939","created_at":"2026-07-05T09:59:59.131537+00:00"},{"alias_kind":"pith_short_12","alias_value":"QRQR2TJZ5KJN","created_at":"2026-07-05T09:59:59.131537+00:00"},{"alias_kind":"pith_short_16","alias_value":"QRQR2TJZ5KJNORLW","created_at":"2026-07-05T09:59:59.131537+00:00"},{"alias_kind":"pith_short_8","alias_value":"QRQR2TJZ","created_at":"2026-07-05T09:59:59.131537+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26734","citing_title":"Robust Onion: Peeling Open Vocab Object Detectors Under Noise","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26734","citing_title":"Robust Onion: Peeling Open Vocab Object Detectors Under Noise","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2602.04476","citing_title":"Vision-aligned Latent Reasoning for Multi-modal Large Language Model","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07914","citing_title":"Mitigating Entangled Steering in Large Vision-Language Models for Hallucination Reduction","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH","json":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH.json","graph_json":"https://pith.science/api/pith-number/QRQR2TJZ5KJNORLW3OVGPX5OSH/graph.json","events_json":"https://pith.science/api/pith-number/QRQR2TJZ5KJNORLW3OVGPX5OSH/events.json","paper":"https://pith.science/paper/QRQR2TJZ"},"agent_actions":{"view_html":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH","download_json":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH.json","view_paper":"https://pith.science/paper/QRQR2TJZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03939&json=true","fetch_graph":"https://pith.science/api/pith-number/QRQR2TJZ5KJNORLW3OVGPX5OSH/graph.json","fetch_events":"https://pith.science/api/pith-number/QRQR2TJZ5KJNORLW3OVGPX5OSH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH/action/storage_attestation","attest_author":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH/action/author_attestation","sign_citation":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH/action/citation_signature","submit_replication":"https://pith.science/pith/QRQR2TJZ5KJNORLW3OVGPX5OSH/action/replication_record"}},"created_at":"2026-07-05T09:59:59.131537+00:00","updated_at":"2026-07-05T09:59:59.131537+00:00"}