{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4RBTI3ULYR54OXKZYPANVJ3VPT","short_pith_number":"pith:4RBTI3UL","schema_version":"1.0","canonical_sha256":"e443346e8bc47bc75d59c3c0daa7757cff539bacc90b2bd03f38902caeb39bc5","source":{"kind":"arxiv","id":"2402.03896","version":3},"attestation_state":"computed","paper":{"title":"Multimodal Rationales for Explainable Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"George Vosselman, Kun Li, Michael Ying Yang","submitted_at":"2024-02-06T11:07:05Z","abstract_excerpt":"Visual Question Answering (VQA) is a challenging task of predicting the answer to a question about the content of an image. Prior works directly evaluate the answering models by simply calculating the accuracy of predicted answers. However, the inner reasoning behind the predictions is disregarded in such a \"black box\" system, and we cannot ascertain the trustworthiness of the predictions. Even more concerning, in some cases, these models predict correct answers despite focusing on irrelevant visual regions or textual tokens. To develop an explainable and trustworthy answering system, we propo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.03896","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-06T11:07:05Z","cross_cats_sorted":[],"title_canon_sha256":"5392bac8841d48191202a269e9c99c3c7aee952783fc70d66e91ef01aae1042d","abstract_canon_sha256":"0a5b5b948f7d05378157866bbd9b6b5bad40cfb5b2aa8ddc379dbaebf75ec212"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:57.721871Z","signature_b64":"kQlq6LOqkeyu/5TRzzHkeyUyY8QNuM39Dfts0gjYlEdM8Lxd5DPCJmGTYYg4TDaGBn/sV0Gt056IZWNBYljzCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e443346e8bc47bc75d59c3c0daa7757cff539bacc90b2bd03f38902caeb39bc5","last_reissued_at":"2026-07-05T11:18:57.721436Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:57.721436Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Rationales for Explainable Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"George Vosselman, Kun Li, Michael Ying Yang","submitted_at":"2024-02-06T11:07:05Z","abstract_excerpt":"Visual Question Answering (VQA) is a challenging task of predicting the answer to a question about the content of an image. Prior works directly evaluate the answering models by simply calculating the accuracy of predicted answers. However, the inner reasoning behind the predictions is disregarded in such a \"black box\" system, and we cannot ascertain the trustworthiness of the predictions. Even more concerning, in some cases, these models predict correct answers despite focusing on irrelevant visual regions or textual tokens. To develop an explainable and trustworthy answering system, we propo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.03896","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.03896/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.03896","created_at":"2026-07-05T11:18:57.721491+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.03896v3","created_at":"2026-07-05T11:18:57.721491+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.03896","created_at":"2026-07-05T11:18:57.721491+00:00"},{"alias_kind":"pith_short_12","alias_value":"4RBTI3ULYR54","created_at":"2026-07-05T11:18:57.721491+00:00"},{"alias_kind":"pith_short_16","alias_value":"4RBTI3ULYR54OXKZ","created_at":"2026-07-05T11:18:57.721491+00:00"},{"alias_kind":"pith_short_8","alias_value":"4RBTI3UL","created_at":"2026-07-05T11:18:57.721491+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.12490","citing_title":"Spatially Grounded Explanations in Vision Language Models for Document Visual Question Answering","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT","json":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT.json","graph_json":"https://pith.science/api/pith-number/4RBTI3ULYR54OXKZYPANVJ3VPT/graph.json","events_json":"https://pith.science/api/pith-number/4RBTI3ULYR54OXKZYPANVJ3VPT/events.json","paper":"https://pith.science/paper/4RBTI3UL"},"agent_actions":{"view_html":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT","download_json":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT.json","view_paper":"https://pith.science/paper/4RBTI3UL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.03896&json=true","fetch_graph":"https://pith.science/api/pith-number/4RBTI3ULYR54OXKZYPANVJ3VPT/graph.json","fetch_events":"https://pith.science/api/pith-number/4RBTI3ULYR54OXKZYPANVJ3VPT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT/action/storage_attestation","attest_author":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT/action/author_attestation","sign_citation":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT/action/citation_signature","submit_replication":"https://pith.science/pith/4RBTI3ULYR54OXKZYPANVJ3VPT/action/replication_record"}},"created_at":"2026-07-05T11:18:57.721491+00:00","updated_at":"2026-07-05T11:18:57.721491+00:00"}