{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TNHMUXAXSV37XMEGJHECPU2PCC","short_pith_number":"pith:TNHMUXAX","schema_version":"1.0","canonical_sha256":"9b4eca5c179577fbb08649c827d34f10b6a3a3910281417f4309ee9008b4902f","source":{"kind":"arxiv","id":"2412.12940","version":1},"attestation_state":"computed","paper":{"title":"Improving Fine-grained Visual Understanding in VLMs through Text-Only Training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dasol Choi, Gio Paik, Guijin Son, Seunghyeok Hong, Soo Yong Kim","submitted_at":"2024-12-17T14:18:50Z","abstract_excerpt":"Visual-Language Models (VLMs) have become a powerful tool for bridging the gap between visual and linguistic understanding. However, the conventional learning approaches for VLMs often suffer from limitations, such as the high resource requirements of collecting and training image-text paired data. Recent research has suggested that language understanding plays a crucial role in the performance of VLMs, potentially indicating that text-only training could be a viable approach. In this work, we investigate the feasibility of enhancing fine-grained visual understanding in VLMs through text-only "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12940","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-17T14:18:50Z","cross_cats_sorted":[],"title_canon_sha256":"82462bc30ec37baf533bc4c6f432181c20e5cf043293fef20e1daf09417ba1e9","abstract_canon_sha256":"b764b261547d6e14526cdbbef8b0277d018b3cfc902b901d6f931ee14d530039"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:51.263994Z","signature_b64":"LUI5PrqoRoNl2NXtbMJlRGPC074JGkFkK4MwAUov4KhlNAwp6zUM/yQGlJW+VoTElHe2YC2Pt20v3DbiEYtKDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b4eca5c179577fbb08649c827d34f10b6a3a3910281417f4309ee9008b4902f","last_reissued_at":"2026-07-05T10:41:51.263542Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:51.263542Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Fine-grained Visual Understanding in VLMs through Text-Only Training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dasol Choi, Gio Paik, Guijin Son, Seunghyeok Hong, Soo Yong Kim","submitted_at":"2024-12-17T14:18:50Z","abstract_excerpt":"Visual-Language Models (VLMs) have become a powerful tool for bridging the gap between visual and linguistic understanding. However, the conventional learning approaches for VLMs often suffer from limitations, such as the high resource requirements of collecting and training image-text paired data. Recent research has suggested that language understanding plays a crucial role in the performance of VLMs, potentially indicating that text-only training could be a viable approach. In this work, we investigate the feasibility of enhancing fine-grained visual understanding in VLMs through text-only "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12940","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12940/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12940","created_at":"2026-07-05T10:41:51.263596+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12940v1","created_at":"2026-07-05T10:41:51.263596+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12940","created_at":"2026-07-05T10:41:51.263596+00:00"},{"alias_kind":"pith_short_12","alias_value":"TNHMUXAXSV37","created_at":"2026-07-05T10:41:51.263596+00:00"},{"alias_kind":"pith_short_16","alias_value":"TNHMUXAXSV37XMEG","created_at":"2026-07-05T10:41:51.263596+00:00"},{"alias_kind":"pith_short_8","alias_value":"TNHMUXAX","created_at":"2026-07-05T10:41:51.263596+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02089","citing_title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20665","citing_title":"The Expense of Seeing: Attaining Trustworthy Multimodal Reasoning Within the Monolithic Paradigm","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20665","citing_title":"The Expense of Seeing: Attaining Trustworthy Multimodal Reasoning Within the Monolithic Paradigm","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC","json":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC.json","graph_json":"https://pith.science/api/pith-number/TNHMUXAXSV37XMEGJHECPU2PCC/graph.json","events_json":"https://pith.science/api/pith-number/TNHMUXAXSV37XMEGJHECPU2PCC/events.json","paper":"https://pith.science/paper/TNHMUXAX"},"agent_actions":{"view_html":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC","download_json":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC.json","view_paper":"https://pith.science/paper/TNHMUXAX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12940&json=true","fetch_graph":"https://pith.science/api/pith-number/TNHMUXAXSV37XMEGJHECPU2PCC/graph.json","fetch_events":"https://pith.science/api/pith-number/TNHMUXAXSV37XMEGJHECPU2PCC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC/action/storage_attestation","attest_author":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC/action/author_attestation","sign_citation":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC/action/citation_signature","submit_replication":"https://pith.science/pith/TNHMUXAXSV37XMEGJHECPU2PCC/action/replication_record"}},"created_at":"2026-07-05T10:41:51.263596+00:00","updated_at":"2026-07-05T10:41:51.263596+00:00"}