{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QHORF3MX5KWYTA4LT47P5S5VUN","short_pith_number":"pith:QHORF3MX","schema_version":"1.0","canonical_sha256":"81dd12ed97eaad89838b9f3efecbb5a3434ceb509bd333a3021b98977534d6a1","source":{"kind":"arxiv","id":"2508.06585","version":2},"attestation_state":"computed","paper":{"title":"CountQA: How Well Do MLLMs Count in the Wild?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Jayant Sravan Tamarapalli, Nilay Pande, Rynaa Grover, Sahiti Yerramilli","submitted_at":"2025-08-08T04:23:04Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) demonstrate remarkable fluency in understanding visual scenes, yet they exhibit a critical lack in a fundamental cognitive skill: object counting. This blind spot severely limits their reliability in real-world applications. To date, this capability has been largely unevaluated in complex scenarios, as existing benchmarks either feature sparse object densities or are confined to specific visual domains, failing to test models under realistic conditions. Addressing this gap, we introduce CountQA, a challenging new benchmark designed to probe this deficie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.06585","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-08T04:23:04Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"53de5bffb8d0222947fea9cfcd240112c00ea22aff1cc478207e5c8101c1150c","abstract_canon_sha256":"74710081fcf78095a308cbfa444e2da0264d4e5be328d8d0dbb4a19d1dd4c5a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:10.227487Z","signature_b64":"3T7GU35YDnTQsoUqqIpN+duIm7b284LELXVh+l+eFONbs/zti5A8DRSF/UAEMaSCRLGO1craYegJ+El+1nr2AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81dd12ed97eaad89838b9f3efecbb5a3434ceb509bd333a3021b98977534d6a1","last_reissued_at":"2026-07-05T12:07:10.226754Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:10.226754Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CountQA: How Well Do MLLMs Count in the Wild?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.AI","authors_text":"Jayant Sravan Tamarapalli, Nilay Pande, Rynaa Grover, Sahiti Yerramilli","submitted_at":"2025-08-08T04:23:04Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) demonstrate remarkable fluency in understanding visual scenes, yet they exhibit a critical lack in a fundamental cognitive skill: object counting. This blind spot severely limits their reliability in real-world applications. To date, this capability has been largely unevaluated in complex scenarios, as existing benchmarks either feature sparse object densities or are confined to specific visual domains, failing to test models under realistic conditions. Addressing this gap, we introduce CountQA, a challenging new benchmark designed to probe this deficie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.06585","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.06585/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.06585","created_at":"2026-07-05T12:07:10.226825+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.06585v2","created_at":"2026-07-05T12:07:10.226825+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.06585","created_at":"2026-07-05T12:07:10.226825+00:00"},{"alias_kind":"pith_short_12","alias_value":"QHORF3MX5KWY","created_at":"2026-07-05T12:07:10.226825+00:00"},{"alias_kind":"pith_short_16","alias_value":"QHORF3MX5KWYTA4L","created_at":"2026-07-05T12:07:10.226825+00:00"},{"alias_kind":"pith_short_8","alias_value":"QHORF3MX","created_at":"2026-07-05T12:07:10.226825+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06420","citing_title":"HoloCount: A Holistic Visual Counting Benchmark for MLLMs","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23835","citing_title":"ABACUS: Adapting Unified Foundation Model for Bridging Image Count Understanding and Generation","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2509.13332","citing_title":"Explicit Reasoning Makes Better Judges: A Systematic Study on Accuracy, Efficiency, and Robustness","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN","json":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN.json","graph_json":"https://pith.science/api/pith-number/QHORF3MX5KWYTA4LT47P5S5VUN/graph.json","events_json":"https://pith.science/api/pith-number/QHORF3MX5KWYTA4LT47P5S5VUN/events.json","paper":"https://pith.science/paper/QHORF3MX"},"agent_actions":{"view_html":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN","download_json":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN.json","view_paper":"https://pith.science/paper/QHORF3MX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.06585&json=true","fetch_graph":"https://pith.science/api/pith-number/QHORF3MX5KWYTA4LT47P5S5VUN/graph.json","fetch_events":"https://pith.science/api/pith-number/QHORF3MX5KWYTA4LT47P5S5VUN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN/action/storage_attestation","attest_author":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN/action/author_attestation","sign_citation":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN/action/citation_signature","submit_replication":"https://pith.science/pith/QHORF3MX5KWYTA4LT47P5S5VUN/action/replication_record"}},"created_at":"2026-07-05T12:07:10.226825+00:00","updated_at":"2026-07-05T12:07:10.226825+00:00"}