{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EE245KREQDRK7CRKOA6F63EY5X","short_pith_number":"pith:EE245KRE","schema_version":"1.0","canonical_sha256":"2135ceaa2480e2af8a2a703c5f6c98edc7ee6c63758a0d69326ecf2d6dbb7326","source":{"kind":"arxiv","id":"2503.14604","version":2},"attestation_state":"computed","paper":{"title":"Image Captioning Evaluation in the Age of Multimodal LLMs: Challenges and Future Perspectives","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Marcella Cornia, Rita Cucchiara, Sara Sarto","submitted_at":"2025-03-18T18:03:56Z","abstract_excerpt":"The evaluation of machine-generated image captions is a complex and evolving challenge. With the advent of Multimodal Large Language Models (MLLMs), image captioning has become a core task, increasing the need for robust and reliable evaluation metrics. This survey provides a comprehensive overview of advancements in image captioning evaluation, analyzing the evolution, strengths, and limitations of existing metrics. We assess these metrics across multiple dimensions, including correlation with human judgment, ranking accuracy, and sensitivity to hallucinations. Additionally, we explore the ch"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.14604","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-18T18:03:56Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"f726deb9f8f7010eeea8883631ab05cad4e9143d91d0d681e4b89fe6249dfc6e","abstract_canon_sha256":"ef93352fbf72f3d478a20f0238e9ecb18997500f39475a23999a565e57edf9ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:29.535616Z","signature_b64":"4jfV6T3MKAAeTt3R+qhgLb2Xyg9A3E2ZNZbryNk1PBzOH86ywFVoZtl0iKUcWzlrKJcKcayT9eYCkq85ImmuCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2135ceaa2480e2af8a2a703c5f6c98edc7ee6c63758a0d69326ecf2d6dbb7326","last_reissued_at":"2026-07-05T11:12:29.535037Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:29.535037Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Image Captioning Evaluation in the Age of Multimodal LLMs: Challenges and Future Perspectives","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Marcella Cornia, Rita Cucchiara, Sara Sarto","submitted_at":"2025-03-18T18:03:56Z","abstract_excerpt":"The evaluation of machine-generated image captions is a complex and evolving challenge. With the advent of Multimodal Large Language Models (MLLMs), image captioning has become a core task, increasing the need for robust and reliable evaluation metrics. This survey provides a comprehensive overview of advancements in image captioning evaluation, analyzing the evolution, strengths, and limitations of existing metrics. We assess these metrics across multiple dimensions, including correlation with human judgment, ranking accuracy, and sensitivity to hallucinations. Additionally, we explore the ch"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.14604","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.14604/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.14604","created_at":"2026-07-05T11:12:29.535109+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.14604v2","created_at":"2026-07-05T11:12:29.535109+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.14604","created_at":"2026-07-05T11:12:29.535109+00:00"},{"alias_kind":"pith_short_12","alias_value":"EE245KREQDRK","created_at":"2026-07-05T11:12:29.535109+00:00"},{"alias_kind":"pith_short_16","alias_value":"EE245KREQDRK7CRK","created_at":"2026-07-05T11:12:29.535109+00:00"},{"alias_kind":"pith_short_8","alias_value":"EE245KRE","created_at":"2026-07-05T11:12:29.535109+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.21692","citing_title":"TCAP: Tri-Component Attention Profiling for Unsupervised Backdoor Detection in MLLM Fine-Tuning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18073","citing_title":"FPBench: A Comprehensive Benchmark of Multimodal Large Language Models for Fingerprint Analysis","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03765","citing_title":"ITIScore: An Image-to-Text-to-Image Rating Framework for the Image Captioning Ability of MLLMs","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10071","citing_title":"Spotlight and Shadow: Attention-Guided Dual-Anchor Introspective Decoding for MLLM Hallucination Mitigation","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X","json":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X.json","graph_json":"https://pith.science/api/pith-number/EE245KREQDRK7CRKOA6F63EY5X/graph.json","events_json":"https://pith.science/api/pith-number/EE245KREQDRK7CRKOA6F63EY5X/events.json","paper":"https://pith.science/paper/EE245KRE"},"agent_actions":{"view_html":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X","download_json":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X.json","view_paper":"https://pith.science/paper/EE245KRE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.14604&json=true","fetch_graph":"https://pith.science/api/pith-number/EE245KREQDRK7CRKOA6F63EY5X/graph.json","fetch_events":"https://pith.science/api/pith-number/EE245KREQDRK7CRKOA6F63EY5X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X/action/storage_attestation","attest_author":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X/action/author_attestation","sign_citation":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X/action/citation_signature","submit_replication":"https://pith.science/pith/EE245KREQDRK7CRKOA6F63EY5X/action/replication_record"}},"created_at":"2026-07-05T11:12:29.535109+00:00","updated_at":"2026-07-05T11:12:29.535109+00:00"}