{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DDZRXRSRMTDXPZBIRV7TFZGVLT","short_pith_number":"pith:DDZRXRSR","schema_version":"1.0","canonical_sha256":"18f31bc65164c777e4288d7f32e4d55cf6223ae907cc9bd5c44afbe725d6d5ba","source":{"kind":"arxiv","id":"2407.17035","version":1},"attestation_state":"computed","paper":{"title":"Q-Ground: Image Quality Grounding with Large Multi-modality Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Annan Wang, Chaofeng Chen, Haoning Wu, Liang Liao, Qiong Yan, Sensen Yang, Weisi Lin, Wenxiu Sun, Zicheng Zhang","submitted_at":"2024-07-24T06:42:46Z","abstract_excerpt":"Recent advances of large multi-modality models (LMM) have greatly improved the ability of image quality assessment (IQA) method to evaluate and explain the quality of visual content. However, these advancements are mostly focused on overall quality assessment, and the detailed examination of local quality, which is crucial for comprehensive visual understanding, is still largely unexplored. In this work, we introduce Q-Ground, the first framework aimed at tackling fine-scale visual quality grounding by combining large multi-modality models with detailed visual quality analysis. Central to our "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.17035","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-24T06:42:46Z","cross_cats_sorted":[],"title_canon_sha256":"bd2725900116c750fa12d573c6943a235dea34b86000ffe34a86930a6a281474","abstract_canon_sha256":"0ea5a415997ccb7c0bcea8fe5dd46c13ededeab9cd456f0af09fd9316f05682a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:01.400565Z","signature_b64":"P80JMToDKWJ3jPKm5bywjI5Swx3EVaUtDEilOtp8crrjSrkRB2QQ/YOsjjXtfxrbjRT3xMZfeVdgLYntszuNAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18f31bc65164c777e4288d7f32e4d55cf6223ae907cc9bd5c44afbe725d6d5ba","last_reissued_at":"2026-07-05T08:48:01.400117Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:01.400117Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Q-Ground: Image Quality Grounding with Large Multi-modality Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Annan Wang, Chaofeng Chen, Haoning Wu, Liang Liao, Qiong Yan, Sensen Yang, Weisi Lin, Wenxiu Sun, Zicheng Zhang","submitted_at":"2024-07-24T06:42:46Z","abstract_excerpt":"Recent advances of large multi-modality models (LMM) have greatly improved the ability of image quality assessment (IQA) method to evaluate and explain the quality of visual content. However, these advancements are mostly focused on overall quality assessment, and the detailed examination of local quality, which is crucial for comprehensive visual understanding, is still largely unexplored. In this work, we introduce Q-Ground, the first framework aimed at tackling fine-scale visual quality grounding by combining large multi-modality models with detailed visual quality analysis. Central to our "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.17035","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.17035/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.17035","created_at":"2026-07-05T08:48:01.400175+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.17035v1","created_at":"2026-07-05T08:48:01.400175+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.17035","created_at":"2026-07-05T08:48:01.400175+00:00"},{"alias_kind":"pith_short_12","alias_value":"DDZRXRSRMTDX","created_at":"2026-07-05T08:48:01.400175+00:00"},{"alias_kind":"pith_short_16","alias_value":"DDZRXRSRMTDXPZBI","created_at":"2026-07-05T08:48:01.400175+00:00"},{"alias_kind":"pith_short_8","alias_value":"DDZRXRSR","created_at":"2026-07-05T08:48:01.400175+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08664","citing_title":"IPAD-CLIP: Teaching CLIP to Detect Image Local Perceptual Artifacts","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT","json":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT.json","graph_json":"https://pith.science/api/pith-number/DDZRXRSRMTDXPZBIRV7TFZGVLT/graph.json","events_json":"https://pith.science/api/pith-number/DDZRXRSRMTDXPZBIRV7TFZGVLT/events.json","paper":"https://pith.science/paper/DDZRXRSR"},"agent_actions":{"view_html":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT","download_json":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT.json","view_paper":"https://pith.science/paper/DDZRXRSR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.17035&json=true","fetch_graph":"https://pith.science/api/pith-number/DDZRXRSRMTDXPZBIRV7TFZGVLT/graph.json","fetch_events":"https://pith.science/api/pith-number/DDZRXRSRMTDXPZBIRV7TFZGVLT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT/action/storage_attestation","attest_author":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT/action/author_attestation","sign_citation":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT/action/citation_signature","submit_replication":"https://pith.science/pith/DDZRXRSRMTDXPZBIRV7TFZGVLT/action/replication_record"}},"created_at":"2026-07-05T08:48:01.400175+00:00","updated_at":"2026-07-05T08:48:01.400175+00:00"}