{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6LIDZBERZUW3BINQRQNSB5TXF6","short_pith_number":"pith:6LIDZBER","schema_version":"1.0","canonical_sha256":"f2d03c8491cd2db0a1b08c1b20f6772f9e152b1efa7d036c833700f491c7ff6b","source":{"kind":"arxiv","id":"2501.09155","version":2},"attestation_state":"computed","paper":{"title":"VCRScore: Image captioning metric based on V\\&L Transformers, CLIP, and precision-recall","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Daniela Moctezuma, Guillermo Ruiz, Tania Ram\\'irez","submitted_at":"2025-01-15T21:14:36Z","abstract_excerpt":"Image captioning has become an essential Vision & Language research task. It is about predicting the most accurate caption given a specific image or video. The research community has achieved impressive results by continuously proposing new models and approaches to improve the overall model's performance. Nevertheless, despite increasing proposals, the performance metrics used to measure their advances have remained practically untouched through the years. A probe of that, nowadays metrics like BLEU, METEOR, CIDEr, and ROUGE are still very used, aside from more sophisticated metrics such as Be"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09155","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-15T21:14:36Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"6459ea2bc386fb1bdbaaa44840817ce0286545e672a09b8d5f45ae1c65aa34f1","abstract_canon_sha256":"839122d535db93bca8dc732458c7e4d80f830c1f35d79ef230d7c31fb2a027d7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:48.256605Z","signature_b64":"cMzegf5ueH2UxaMktS9j0tw717hMURl/+bPQWTEJ9rlWSVCyWbp5+FYuuhKH1e+pHbNxkuEnbhgluRSOUkbrCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2d03c8491cd2db0a1b08c1b20f6772f9e152b1efa7d036c833700f491c7ff6b","last_reissued_at":"2026-07-05T10:05:48.256104Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:48.256104Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VCRScore: Image captioning metric based on V\\&L Transformers, CLIP, and precision-recall","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Daniela Moctezuma, Guillermo Ruiz, Tania Ram\\'irez","submitted_at":"2025-01-15T21:14:36Z","abstract_excerpt":"Image captioning has become an essential Vision & Language research task. It is about predicting the most accurate caption given a specific image or video. The research community has achieved impressive results by continuously proposing new models and approaches to improve the overall model's performance. Nevertheless, despite increasing proposals, the performance metrics used to measure their advances have remained practically untouched through the years. A probe of that, nowadays metrics like BLEU, METEOR, CIDEr, and ROUGE are still very used, aside from more sophisticated metrics such as Be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09155","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09155/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09155","created_at":"2026-07-05T10:05:48.256163+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09155v2","created_at":"2026-07-05T10:05:48.256163+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09155","created_at":"2026-07-05T10:05:48.256163+00:00"},{"alias_kind":"pith_short_12","alias_value":"6LIDZBERZUW3","created_at":"2026-07-05T10:05:48.256163+00:00"},{"alias_kind":"pith_short_16","alias_value":"6LIDZBERZUW3BINQ","created_at":"2026-07-05T10:05:48.256163+00:00"},{"alias_kind":"pith_short_8","alias_value":"6LIDZBER","created_at":"2026-07-05T10:05:48.256163+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6","json":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6.json","graph_json":"https://pith.science/api/pith-number/6LIDZBERZUW3BINQRQNSB5TXF6/graph.json","events_json":"https://pith.science/api/pith-number/6LIDZBERZUW3BINQRQNSB5TXF6/events.json","paper":"https://pith.science/paper/6LIDZBER"},"agent_actions":{"view_html":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6","download_json":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6.json","view_paper":"https://pith.science/paper/6LIDZBER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09155&json=true","fetch_graph":"https://pith.science/api/pith-number/6LIDZBERZUW3BINQRQNSB5TXF6/graph.json","fetch_events":"https://pith.science/api/pith-number/6LIDZBERZUW3BINQRQNSB5TXF6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6/action/storage_attestation","attest_author":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6/action/author_attestation","sign_citation":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6/action/citation_signature","submit_replication":"https://pith.science/pith/6LIDZBERZUW3BINQRQNSB5TXF6/action/replication_record"}},"created_at":"2026-07-05T10:05:48.256163+00:00","updated_at":"2026-07-05T10:05:48.256163+00:00"}