{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:AJML3EYMX6LEHW4WS4NI3D6GNQ","short_pith_number":"pith:AJML3EYM","schema_version":"1.0","canonical_sha256":"0258bd930cbf9643db96971a8d8fc66c3c0bc89b18c01b9e4fbabc6d4894dcdb","source":{"kind":"arxiv","id":"2607.06420","version":1},"attestation_state":"computed","paper":{"title":"HoloCount: A Holistic Visual Counting Benchmark for MLLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guanglu Wan, Jinhong Deng, Limeng Qiao","submitted_at":"2026-07-07T15:48:57Z","abstract_excerpt":"Visual counting is a fundamental pillar of multimodal intelligence, requiring a seamless integration of fine-grained grounding and spatial reasoning. While Multimodal Large Language Models (MLLMs) have achieved remarkable success in qualitative scene understanding, their quantitative precision remains a significant bottleneck, often characterized by persistent numerical hallucinations. Existing counting benchmarks primarily focus on basic perception in simplified contexts, failing to capture the complex failure modes that emerge under logical constraints or adversarial conditions. To address t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.06420","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-07-07T15:48:57Z","cross_cats_sorted":[],"title_canon_sha256":"5bfd8f4f35b28859e495639fcdd8e407b240b5b1338f97ec69f6655d583c69a6","abstract_canon_sha256":"2ce0585ef055eb919b6ec43448184ecce6e88c5091c5701c7293135e30c5b712"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-08T01:19:24.843753Z","signature_b64":"8U9yTm6E/WJSjAWt1GQUtSAPFQYc49SfZSqK3eUmoD1BmuHrDko+/RMoTJw81fStHcyWr+7luJyXOTCUrDRhCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0258bd930cbf9643db96971a8d8fc66c3c0bc89b18c01b9e4fbabc6d4894dcdb","last_reissued_at":"2026-07-08T01:19:24.843344Z","signature_status":"signed_v1","first_computed_at":"2026-07-08T01:19:24.843344Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HoloCount: A Holistic Visual Counting Benchmark for MLLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guanglu Wan, Jinhong Deng, Limeng Qiao","submitted_at":"2026-07-07T15:48:57Z","abstract_excerpt":"Visual counting is a fundamental pillar of multimodal intelligence, requiring a seamless integration of fine-grained grounding and spatial reasoning. While Multimodal Large Language Models (MLLMs) have achieved remarkable success in qualitative scene understanding, their quantitative precision remains a significant bottleneck, often characterized by persistent numerical hallucinations. Existing counting benchmarks primarily focus on basic perception in simplified contexts, failing to capture the complex failure modes that emerge under logical constraints or adversarial conditions. To address t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.06420","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.06420/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.06420","created_at":"2026-07-08T01:19:24.843395+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.06420v1","created_at":"2026-07-08T01:19:24.843395+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.06420","created_at":"2026-07-08T01:19:24.843395+00:00"},{"alias_kind":"pith_short_12","alias_value":"AJML3EYMX6LE","created_at":"2026-07-08T01:19:24.843395+00:00"},{"alias_kind":"pith_short_16","alias_value":"AJML3EYMX6LEHW4W","created_at":"2026-07-08T01:19:24.843395+00:00"},{"alias_kind":"pith_short_8","alias_value":"AJML3EYM","created_at":"2026-07-08T01:19:24.843395+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ","json":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ.json","graph_json":"https://pith.science/api/pith-number/AJML3EYMX6LEHW4WS4NI3D6GNQ/graph.json","events_json":"https://pith.science/api/pith-number/AJML3EYMX6LEHW4WS4NI3D6GNQ/events.json","paper":"https://pith.science/paper/AJML3EYM"},"agent_actions":{"view_html":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ","download_json":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ.json","view_paper":"https://pith.science/paper/AJML3EYM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.06420&json=true","fetch_graph":"https://pith.science/api/pith-number/AJML3EYMX6LEHW4WS4NI3D6GNQ/graph.json","fetch_events":"https://pith.science/api/pith-number/AJML3EYMX6LEHW4WS4NI3D6GNQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ/action/storage_attestation","attest_author":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ/action/author_attestation","sign_citation":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ/action/citation_signature","submit_replication":"https://pith.science/pith/AJML3EYMX6LEHW4WS4NI3D6GNQ/action/replication_record"}},"created_at":"2026-07-08T01:19:24.843395+00:00","updated_at":"2026-07-08T01:19:24.843395+00:00"}