{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZKJMBYAS43K65I6NVNXRNMNGZN","short_pith_number":"pith:ZKJMBYAS","schema_version":"1.0","canonical_sha256":"ca92c0e012e6d5eea3cdab6f16b1a6cb50ad0116dd7d428295a0d83f9f9801ff","source":{"kind":"arxiv","id":"2504.14848","version":1},"attestation_state":"computed","paper":{"title":"Object-Level Verbalized Confidence Calibration in Vision-Language Models via Semantic Perturbation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jiaming Guo, Junbin Xiao, Ruibo Hou, Rui Zhang, Yifan Hao, Yunji Chen, Yunpu Zhao, Zihao Zhang","submitted_at":"2025-04-21T04:01:22Z","abstract_excerpt":"Vision-language models (VLMs) excel in various multimodal tasks but frequently suffer from poor calibration, resulting in misalignment between their verbalized confidence and response correctness. This miscalibration undermines user trust, especially when models confidently provide incorrect or fabricated information. In this work, we propose a novel Confidence Calibration through Semantic Perturbation (CSP) framework to improve the calibration of verbalized confidence for VLMs in response to object-centric queries. We first introduce a perturbed dataset where Gaussian noise is applied to the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.14848","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-21T04:01:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"61882f6f54dec1e8d01e3760ef61ca72088b85b3e1c6ae20bb50897307c765d6","abstract_canon_sha256":"264abab3129e6c17c13291ff53480de62e6397f14ceac37ae7385772b6ec5635"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:51.547726Z","signature_b64":"z3RIKCuE8lSEk7RVj90VRCB8ajl9WEw57BO3xFrZeAZTLIEJB7jaov9OBc2JTlF9PQoEgXSwsOvEiISCjoRKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca92c0e012e6d5eea3cdab6f16b1a6cb50ad0116dd7d428295a0d83f9f9801ff","last_reissued_at":"2026-07-05T10:51:51.547217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:51.547217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Object-Level Verbalized Confidence Calibration in Vision-Language Models via Semantic Perturbation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jiaming Guo, Junbin Xiao, Ruibo Hou, Rui Zhang, Yifan Hao, Yunji Chen, Yunpu Zhao, Zihao Zhang","submitted_at":"2025-04-21T04:01:22Z","abstract_excerpt":"Vision-language models (VLMs) excel in various multimodal tasks but frequently suffer from poor calibration, resulting in misalignment between their verbalized confidence and response correctness. This miscalibration undermines user trust, especially when models confidently provide incorrect or fabricated information. In this work, we propose a novel Confidence Calibration through Semantic Perturbation (CSP) framework to improve the calibration of verbalized confidence for VLMs in response to object-centric queries. We first introduce a perturbed dataset where Gaussian noise is applied to the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14848","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14848/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.14848","created_at":"2026-07-05T10:51:51.547281+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.14848v1","created_at":"2026-07-05T10:51:51.547281+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14848","created_at":"2026-07-05T10:51:51.547281+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZKJMBYAS43K6","created_at":"2026-07-05T10:51:51.547281+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZKJMBYAS43K65I6N","created_at":"2026-07-05T10:51:51.547281+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZKJMBYAS","created_at":"2026-07-05T10:51:51.547281+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10893","citing_title":"Grounded or Guessing? LVLM Confidence Estimation via Blind-Image Contrastive Ranking","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10893","citing_title":"Grounded or Guessing? LVLM Confidence Estimation via Blind-Image Contrastive Ranking","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN","json":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN.json","graph_json":"https://pith.science/api/pith-number/ZKJMBYAS43K65I6NVNXRNMNGZN/graph.json","events_json":"https://pith.science/api/pith-number/ZKJMBYAS43K65I6NVNXRNMNGZN/events.json","paper":"https://pith.science/paper/ZKJMBYAS"},"agent_actions":{"view_html":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN","download_json":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN.json","view_paper":"https://pith.science/paper/ZKJMBYAS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.14848&json=true","fetch_graph":"https://pith.science/api/pith-number/ZKJMBYAS43K65I6NVNXRNMNGZN/graph.json","fetch_events":"https://pith.science/api/pith-number/ZKJMBYAS43K65I6NVNXRNMNGZN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN/action/storage_attestation","attest_author":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN/action/author_attestation","sign_citation":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN/action/citation_signature","submit_replication":"https://pith.science/pith/ZKJMBYAS43K65I6NVNXRNMNGZN/action/replication_record"}},"created_at":"2026-07-05T10:51:51.547281+00:00","updated_at":"2026-07-05T10:51:51.547281+00:00"}