{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DKGJFUYEBQLAW4R5VZAGDPS2LP","short_pith_number":"pith:DKGJFUYE","schema_version":"1.0","canonical_sha256":"1a8c92d3040c160b723dae4061be5a5bd7aa190841d1d61a323dbe52c0e2b1de","source":{"kind":"arxiv","id":"2508.02645","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Variance in Visual Question Answering Benchmarks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Nikitha SR","submitted_at":"2025-08-04T17:37:13Z","abstract_excerpt":"Multimodal large language models (MLLMs) have emerged as powerful tools for visual question answering (VQA), enabling reasoning and contextual understanding across visual and textual modalities. Despite their advancements, the evaluation of MLLMs on VQA benchmarks often relies on point estimates, overlooking the significant variance in performance caused by factors such as stochastic model outputs, training seed sensitivity, and hyperparameter configurations. This paper critically examines these issues by analyzing variance across 14 widely used VQA benchmarks, covering diverse tasks such as v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.02645","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-04T17:37:13Z","cross_cats_sorted":[],"title_canon_sha256":"30e09cfecc5ae2fcb23b9c24a33aba81aa9d7c7bc4c529f4e3ce1da053ab29d6","abstract_canon_sha256":"0b403a56f809f413a0bfd43e84cbef2cdc7547579e4f7c7886d4770c22053fca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:17.850780Z","signature_b64":"6mr/VTct4SmuZWsi5uytd2TrUUUUEK3suphn72ltyvxuVGDSBdV1YXxJs/L6Uiqd5s8K4bIOdWGJiSplUHs3CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a8c92d3040c160b723dae4061be5a5bd7aa190841d1d61a323dbe52c0e2b1de","last_reissued_at":"2026-07-05T11:48:17.850270Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:17.850270Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Variance in Visual Question Answering Benchmarks","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Nikitha SR","submitted_at":"2025-08-04T17:37:13Z","abstract_excerpt":"Multimodal large language models (MLLMs) have emerged as powerful tools for visual question answering (VQA), enabling reasoning and contextual understanding across visual and textual modalities. Despite their advancements, the evaluation of MLLMs on VQA benchmarks often relies on point estimates, overlooking the significant variance in performance caused by factors such as stochastic model outputs, training seed sensitivity, and hyperparameter configurations. This paper critically examines these issues by analyzing variance across 14 widely used VQA benchmarks, covering diverse tasks such as v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.02645","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.02645/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.02645","created_at":"2026-07-05T11:48:17.850336+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.02645v1","created_at":"2026-07-05T11:48:17.850336+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.02645","created_at":"2026-07-05T11:48:17.850336+00:00"},{"alias_kind":"pith_short_12","alias_value":"DKGJFUYEBQLA","created_at":"2026-07-05T11:48:17.850336+00:00"},{"alias_kind":"pith_short_16","alias_value":"DKGJFUYEBQLAW4R5","created_at":"2026-07-05T11:48:17.850336+00:00"},{"alias_kind":"pith_short_8","alias_value":"DKGJFUYE","created_at":"2026-07-05T11:48:17.850336+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP","json":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP.json","graph_json":"https://pith.science/api/pith-number/DKGJFUYEBQLAW4R5VZAGDPS2LP/graph.json","events_json":"https://pith.science/api/pith-number/DKGJFUYEBQLAW4R5VZAGDPS2LP/events.json","paper":"https://pith.science/paper/DKGJFUYE"},"agent_actions":{"view_html":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP","download_json":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP.json","view_paper":"https://pith.science/paper/DKGJFUYE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.02645&json=true","fetch_graph":"https://pith.science/api/pith-number/DKGJFUYEBQLAW4R5VZAGDPS2LP/graph.json","fetch_events":"https://pith.science/api/pith-number/DKGJFUYEBQLAW4R5VZAGDPS2LP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP/action/storage_attestation","attest_author":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP/action/author_attestation","sign_citation":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP/action/citation_signature","submit_replication":"https://pith.science/pith/DKGJFUYEBQLAW4R5VZAGDPS2LP/action/replication_record"}},"created_at":"2026-07-05T11:48:17.850336+00:00","updated_at":"2026-07-05T11:48:17.850336+00:00"}