{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7NGKZ2GTGF2BNFTB2CKPGWRISG","short_pith_number":"pith:7NGKZ2GT","schema_version":"1.0","canonical_sha256":"fb4cace8d33174169661d094f35a2891a67387b695e66d95aa2a37528913ee59","source":{"kind":"arxiv","id":"2505.22613","version":1},"attestation_state":"computed","paper":{"title":"RICO: Improving Accuracy and Completeness in Image Recaptioning via Visual Reconstruction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Linli Yao, Pengfei Wan, Shuhuai Ren, Sihan Yang, Xu Sun, Yishuo Cai, Yuanxing Zhang, Yuanxin Liu, Yuchi Wang","submitted_at":"2025-05-28T17:29:34Z","abstract_excerpt":"Image recaptioning is widely used to generate training datasets with enhanced quality for various multimodal tasks. Existing recaptioning methods typically rely on powerful multimodal large language models (MLLMs) to enhance textual descriptions, but often suffer from inaccuracies due to hallucinations and incompleteness caused by missing fine-grained details. To address these limitations, we propose RICO, a novel framework that refines captions through visual reconstruction. Specifically, we leverage a text-to-image model to reconstruct a caption into a reference image, and prompt an MLLM to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22613","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-28T17:29:34Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"1995d02e418cd333febe5254cf856ff7706d765ee858cab00aa99f477aa17f88","abstract_canon_sha256":"72a8eb43b4a762b5bfc4c30dcf4c011cfd73b89ad4786c28ff8ff137188316ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:27.834790Z","signature_b64":"/3AmCWqGyCjp7bMkFLHz7sljaEBnfo+CzGE0/phRRfnnpYB6b0axSltYXJSS+MWDK7HtWeAvDzXJdFMpcfK9DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb4cace8d33174169661d094f35a2891a67387b695e66d95aa2a37528913ee59","last_reissued_at":"2026-07-05T11:11:27.834278Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:27.834278Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RICO: Improving Accuracy and Completeness in Image Recaptioning via Visual Reconstruction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Linli Yao, Pengfei Wan, Shuhuai Ren, Sihan Yang, Xu Sun, Yishuo Cai, Yuanxing Zhang, Yuanxin Liu, Yuchi Wang","submitted_at":"2025-05-28T17:29:34Z","abstract_excerpt":"Image recaptioning is widely used to generate training datasets with enhanced quality for various multimodal tasks. Existing recaptioning methods typically rely on powerful multimodal large language models (MLLMs) to enhance textual descriptions, but often suffer from inaccuracies due to hallucinations and incompleteness caused by missing fine-grained details. To address these limitations, we propose RICO, a novel framework that refines captions through visual reconstruction. Specifically, we leverage a text-to-image model to reconstruct a caption into a reference image, and prompt an MLLM to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22613","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22613/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22613","created_at":"2026-07-05T11:11:27.834344+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22613v1","created_at":"2026-07-05T11:11:27.834344+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22613","created_at":"2026-07-05T11:11:27.834344+00:00"},{"alias_kind":"pith_short_12","alias_value":"7NGKZ2GTGF2B","created_at":"2026-07-05T11:11:27.834344+00:00"},{"alias_kind":"pith_short_16","alias_value":"7NGKZ2GTGF2BNFTB","created_at":"2026-07-05T11:11:27.834344+00:00"},{"alias_kind":"pith_short_8","alias_value":"7NGKZ2GT","created_at":"2026-07-05T11:11:27.834344+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG","json":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG.json","graph_json":"https://pith.science/api/pith-number/7NGKZ2GTGF2BNFTB2CKPGWRISG/graph.json","events_json":"https://pith.science/api/pith-number/7NGKZ2GTGF2BNFTB2CKPGWRISG/events.json","paper":"https://pith.science/paper/7NGKZ2GT"},"agent_actions":{"view_html":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG","download_json":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG.json","view_paper":"https://pith.science/paper/7NGKZ2GT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22613&json=true","fetch_graph":"https://pith.science/api/pith-number/7NGKZ2GTGF2BNFTB2CKPGWRISG/graph.json","fetch_events":"https://pith.science/api/pith-number/7NGKZ2GTGF2BNFTB2CKPGWRISG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG/action/storage_attestation","attest_author":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG/action/author_attestation","sign_citation":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG/action/citation_signature","submit_replication":"https://pith.science/pith/7NGKZ2GTGF2BNFTB2CKPGWRISG/action/replication_record"}},"created_at":"2026-07-05T11:11:27.834344+00:00","updated_at":"2026-07-05T11:11:27.834344+00:00"}