{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YLI236GCBPUML3F3L6EBJH7KCJ","short_pith_number":"pith:YLI236GC","schema_version":"1.0","canonical_sha256":"c2d1adf8c20be8c5ecbb5f88149fea126c2d09451b7d98afa225cb38f05e97c0","source":{"kind":"arxiv","id":"2411.18203","version":5},"attestation_state":"computed","paper":{"title":"Critic-V: VLM Critics Help Catch VLM Errors in Multimodal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Di Zhang, Dongzhan Zhou, Jianbo Wu, Jiatong Li, Jingdi Lei, Junxian Li, Peng Ye, Suorong Yang, Wanli Ouyang, Weida Wang, Xunzhi Wang, Yujie Liu, Zonglin Yang","submitted_at":"2024-11-27T10:28:57Z","abstract_excerpt":"Vision-language models (VLMs) have shown remarkable advancements in multimodal reasoning tasks. However, they still often generate inaccurate or irrelevant responses due to issues like hallucinated image understandings or unrefined reasoning paths. To address these challenges, we introduce Critic-V, a novel framework inspired by the Actor-Critic paradigm to boost the reasoning capability of VLMs. This framework decouples the reasoning process and critic process by integrating two independent components: the Reasoner, which generates reasoning paths based on visual and textual inputs, and the C"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.18203","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-27T10:28:57Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"14f6a7ec5e77d046b3295800cf179d8b4a401768718510cd4e2bed0d13631574","abstract_canon_sha256":"332b2c3b28d2bf0a8b421ae998541759a64421858ffdde3ec171b07eedc0b9e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:56.787416Z","signature_b64":"n5PuNm78LNKaXBVPZoiZiCayIx2mDG7X++r3bMRhT2lBwUtqW6h8zQLiqF1Ev9o0WDMCa5KxST1HUOp3SnsGAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2d1adf8c20be8c5ecbb5f88149fea126c2d09451b7d98afa225cb38f05e97c0","last_reissued_at":"2026-07-05T10:52:56.786899Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:56.786899Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Critic-V: VLM Critics Help Catch VLM Errors in Multimodal Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Di Zhang, Dongzhan Zhou, Jianbo Wu, Jiatong Li, Jingdi Lei, Junxian Li, Peng Ye, Suorong Yang, Wanli Ouyang, Weida Wang, Xunzhi Wang, Yujie Liu, Zonglin Yang","submitted_at":"2024-11-27T10:28:57Z","abstract_excerpt":"Vision-language models (VLMs) have shown remarkable advancements in multimodal reasoning tasks. However, they still often generate inaccurate or irrelevant responses due to issues like hallucinated image understandings or unrefined reasoning paths. To address these challenges, we introduce Critic-V, a novel framework inspired by the Actor-Critic paradigm to boost the reasoning capability of VLMs. This framework decouples the reasoning process and critic process by integrating two independent components: the Reasoner, which generates reasoning paths based on visual and textual inputs, and the C"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.18203","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.18203/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.18203","created_at":"2026-07-05T10:52:56.786959+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.18203v5","created_at":"2026-07-05T10:52:56.786959+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.18203","created_at":"2026-07-05T10:52:56.786959+00:00"},{"alias_kind":"pith_short_12","alias_value":"YLI236GCBPUM","created_at":"2026-07-05T10:52:56.786959+00:00"},{"alias_kind":"pith_short_16","alias_value":"YLI236GCBPUML3F3","created_at":"2026-07-05T10:52:56.786959+00:00"},{"alias_kind":"pith_short_8","alias_value":"YLI236GC","created_at":"2026-07-05T10:52:56.786959+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01671","citing_title":"When Meaning Travels: A Granular Lens on Hybrid-MoE's Role in Idiomatic Understanding for Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16410","citing_title":"Test-Time Hinting for Black-Box Vision-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":173,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ","json":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ.json","graph_json":"https://pith.science/api/pith-number/YLI236GCBPUML3F3L6EBJH7KCJ/graph.json","events_json":"https://pith.science/api/pith-number/YLI236GCBPUML3F3L6EBJH7KCJ/events.json","paper":"https://pith.science/paper/YLI236GC"},"agent_actions":{"view_html":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ","download_json":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ.json","view_paper":"https://pith.science/paper/YLI236GC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.18203&json=true","fetch_graph":"https://pith.science/api/pith-number/YLI236GCBPUML3F3L6EBJH7KCJ/graph.json","fetch_events":"https://pith.science/api/pith-number/YLI236GCBPUML3F3L6EBJH7KCJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ/action/storage_attestation","attest_author":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ/action/author_attestation","sign_citation":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ/action/citation_signature","submit_replication":"https://pith.science/pith/YLI236GCBPUML3F3L6EBJH7KCJ/action/replication_record"}},"created_at":"2026-07-05T10:52:56.786959+00:00","updated_at":"2026-07-05T10:52:56.786959+00:00"}