{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FGKB4CL5Q4IR2BYEBVGKOLZ6PD","short_pith_number":"pith:FGKB4CL5","schema_version":"1.0","canonical_sha256":"29941e097d87111d07040d4ca72f3e78e18be9b79bd79fdeb8b1725d3877006a","source":{"kind":"arxiv","id":"2411.15411","version":1},"attestation_state":"computed","paper":{"title":"FINECAPTION: Compositional Image Captioning Focusing on Wherever You Want at Any Granularity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Hua, Jianming Zhang, Jiebo Luo, Jing Shi, Lingzhi Zhang, Qing Liu, Yilin Wang, Zhifei Zhang","submitted_at":"2024-11-23T02:20:32Z","abstract_excerpt":"The advent of large Vision-Language Models (VLMs) has significantly advanced multimodal tasks, enabling more sophisticated and accurate reasoning across various applications, including image and video captioning, visual question answering, and cross-modal retrieval. Despite their superior capabilities, VLMs struggle with fine-grained image regional composition information perception. Specifically, they have difficulty accurately aligning the segmentation masks with the corresponding semantics and precisely describing the compositional aspects of the referred regions.\n  However, compositionalit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15411","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-23T02:20:32Z","cross_cats_sorted":[],"title_canon_sha256":"e6d7836bd7f6004b06cffd729c4fb9a7001391933e41deb877d407d31fa93042","abstract_canon_sha256":"023a208f8d06414ecf709c545cf0a10bac5cc1c7ad7f0ed19ba2aa97c43a66f3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:34.170522Z","signature_b64":"hdeF3DCppZEIsjKLRyoBxN0DYbkuEj15bR8axo/sM40bBqvVGXNHoipfuE/2z+fCGGy2wPlUOqztgq7zyEjLAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29941e097d87111d07040d4ca72f3e78e18be9b79bd79fdeb8b1725d3877006a","last_reissued_at":"2026-07-05T09:39:34.170063Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:34.170063Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FINECAPTION: Compositional Image Captioning Focusing on Wherever You Want at Any Granularity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Hua, Jianming Zhang, Jiebo Luo, Jing Shi, Lingzhi Zhang, Qing Liu, Yilin Wang, Zhifei Zhang","submitted_at":"2024-11-23T02:20:32Z","abstract_excerpt":"The advent of large Vision-Language Models (VLMs) has significantly advanced multimodal tasks, enabling more sophisticated and accurate reasoning across various applications, including image and video captioning, visual question answering, and cross-modal retrieval. Despite their superior capabilities, VLMs struggle with fine-grained image regional composition information perception. Specifically, they have difficulty accurately aligning the segmentation masks with the corresponding semantics and precisely describing the compositional aspects of the referred regions.\n  However, compositionalit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15411","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15411/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15411","created_at":"2026-07-05T09:39:34.170118+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15411v1","created_at":"2026-07-05T09:39:34.170118+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15411","created_at":"2026-07-05T09:39:34.170118+00:00"},{"alias_kind":"pith_short_12","alias_value":"FGKB4CL5Q4IR","created_at":"2026-07-05T09:39:34.170118+00:00"},{"alias_kind":"pith_short_16","alias_value":"FGKB4CL5Q4IR2BYE","created_at":"2026-07-05T09:39:34.170118+00:00"},{"alias_kind":"pith_short_8","alias_value":"FGKB4CL5","created_at":"2026-07-05T09:39:34.170118+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.22627","citing_title":"Chain-of-Talkers (CoTalk): Fast Human Annotation of Dense Image Captions","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD","json":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD.json","graph_json":"https://pith.science/api/pith-number/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/graph.json","events_json":"https://pith.science/api/pith-number/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/events.json","paper":"https://pith.science/paper/FGKB4CL5"},"agent_actions":{"view_html":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD","download_json":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD.json","view_paper":"https://pith.science/paper/FGKB4CL5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15411&json=true","fetch_graph":"https://pith.science/api/pith-number/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/graph.json","fetch_events":"https://pith.science/api/pith-number/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/action/storage_attestation","attest_author":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/action/author_attestation","sign_citation":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/action/citation_signature","submit_replication":"https://pith.science/pith/FGKB4CL5Q4IR2BYEBVGKOLZ6PD/action/replication_record"}},"created_at":"2026-07-05T09:39:34.170118+00:00","updated_at":"2026-07-05T09:39:34.170118+00:00"}