{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F466FGMCIDFW7IUCFQP3NE7PYF","short_pith_number":"pith:F466FGMC","schema_version":"1.0","canonical_sha256":"2f3de2998240cb6fa2822c1fb693efc147f05c4c52edfe994885abb1d52e3dd3","source":{"kind":"arxiv","id":"2409.15125","version":1},"attestation_state":"computed","paper":{"title":"Detect, Describe, Discriminate: Moving Beyond VQA for MLLM Evaluation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Darshan Singh S, Makarand Tapaswi, Manu Gaur","submitted_at":"2024-09-23T15:31:25Z","abstract_excerpt":"Visual Question Answering (VQA) with multiple choice questions enables a vision-centric evaluation of Multimodal Large Language Models (MLLMs). Although it reliably checks the existence of specific visual abilities, it is easier for the model to select an answer from multiple choices (VQA evaluation) than to generate the answer itself. In this work, we offer a novel perspective: we evaluate how well an MLLM understands a specific visual concept by its ability to uniquely describe two extremely similar images that differ only in the targeted visual concept. Specifically, we assess the ability o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.15125","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-23T15:31:25Z","cross_cats_sorted":[],"title_canon_sha256":"42489ff03f0f90bb5d5577c4cdd8d3782a16cf106f86210a2f19f9bf8b03d88a","abstract_canon_sha256":"2580a2706115467cecdf32fb90500da692bbbdc6530170691134027873590a04"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:36.568269Z","signature_b64":"4kEFju3bESs9tNiwVrYM8oo45qNtUOp82LKTjo3hLLJVayv6TBVl+Kvrjk/JVPfj3P1yk+yqb+E7vvzHwxVwDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f3de2998240cb6fa2822c1fb693efc147f05c4c52edfe994885abb1d52e3dd3","last_reissued_at":"2026-07-05T09:10:36.567859Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:36.567859Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Detect, Describe, Discriminate: Moving Beyond VQA for MLLM Evaluation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Darshan Singh S, Makarand Tapaswi, Manu Gaur","submitted_at":"2024-09-23T15:31:25Z","abstract_excerpt":"Visual Question Answering (VQA) with multiple choice questions enables a vision-centric evaluation of Multimodal Large Language Models (MLLMs). Although it reliably checks the existence of specific visual abilities, it is easier for the model to select an answer from multiple choices (VQA evaluation) than to generate the answer itself. In this work, we offer a novel perspective: we evaluate how well an MLLM understands a specific visual concept by its ability to uniquely describe two extremely similar images that differ only in the targeted visual concept. Specifically, we assess the ability o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15125","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.15125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.15125","created_at":"2026-07-05T09:10:36.567918+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.15125v1","created_at":"2026-07-05T09:10:36.567918+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15125","created_at":"2026-07-05T09:10:36.567918+00:00"},{"alias_kind":"pith_short_12","alias_value":"F466FGMCIDFW","created_at":"2026-07-05T09:10:36.567918+00:00"},{"alias_kind":"pith_short_16","alias_value":"F466FGMCIDFW7IUC","created_at":"2026-07-05T09:10:36.567918+00:00"},{"alias_kind":"pith_short_8","alias_value":"F466FGMC","created_at":"2026-07-05T09:10:36.567918+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF","json":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF.json","graph_json":"https://pith.science/api/pith-number/F466FGMCIDFW7IUCFQP3NE7PYF/graph.json","events_json":"https://pith.science/api/pith-number/F466FGMCIDFW7IUCFQP3NE7PYF/events.json","paper":"https://pith.science/paper/F466FGMC"},"agent_actions":{"view_html":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF","download_json":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF.json","view_paper":"https://pith.science/paper/F466FGMC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.15125&json=true","fetch_graph":"https://pith.science/api/pith-number/F466FGMCIDFW7IUCFQP3NE7PYF/graph.json","fetch_events":"https://pith.science/api/pith-number/F466FGMCIDFW7IUCFQP3NE7PYF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF/action/storage_attestation","attest_author":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF/action/author_attestation","sign_citation":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF/action/citation_signature","submit_replication":"https://pith.science/pith/F466FGMCIDFW7IUCFQP3NE7PYF/action/replication_record"}},"created_at":"2026-07-05T09:10:36.567918+00:00","updated_at":"2026-07-05T09:10:36.567918+00:00"}