{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RESASOFFZHDIKECM7KONTENQBD","short_pith_number":"pith:RESASOFF","schema_version":"1.0","canonical_sha256":"89240938a5c9c685104cfa9cd991b008f4c587e9b1b571ec4a400e8d20c73754","source":{"kind":"arxiv","id":"2508.02886","version":1},"attestation_state":"computed","paper":{"title":"Coherent Multimodal Reasoning with Iterative Self-Evaluation for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Julian Perry, Ruocheng Li, Shanshan Zhu, Wenjie Luo","submitted_at":"2025-08-04T20:33:58Z","abstract_excerpt":"Despite significant advancements, current large language models (LLMs) and vision-language models (LVLMs) continue to struggle with complex, multi-step, cross-modal common sense reasoning tasks, often exhibiting a lack of \"deliberative thinking.\" They tend to rely on superficial associations rather than deep, chained inference, particularly when integrating visual information with abstract concepts. To address this, we propose the Coherent Multimodal Reasoning Framework (CMRF), a novel approach that enhances LVLMs' common sense reasoning capabilities through an iterative, self-evaluating infer"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.02886","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-04T20:33:58Z","cross_cats_sorted":[],"title_canon_sha256":"4dc12367588109361ece5ad2d1901c2d44583f4a95ec54e770e44b4101176bb9","abstract_canon_sha256":"e11cdc5a3d6ec5794b6e735a5df99baac4273ceb5ae3f1a392225caf9bc6a758"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:32.338893Z","signature_b64":"BP2ZqVRADF6j+qCtSublB6btVuUKUSdxZwlK05MnkSdkII1HARQJ6tCACKtaUzW+ry0lmVAME+dicTcAmd1ACg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89240938a5c9c685104cfa9cd991b008f4c587e9b1b571ec4a400e8d20c73754","last_reissued_at":"2026-07-05T11:48:32.338371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:32.338371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Coherent Multimodal Reasoning with Iterative Self-Evaluation for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Julian Perry, Ruocheng Li, Shanshan Zhu, Wenjie Luo","submitted_at":"2025-08-04T20:33:58Z","abstract_excerpt":"Despite significant advancements, current large language models (LLMs) and vision-language models (LVLMs) continue to struggle with complex, multi-step, cross-modal common sense reasoning tasks, often exhibiting a lack of \"deliberative thinking.\" They tend to rely on superficial associations rather than deep, chained inference, particularly when integrating visual information with abstract concepts. To address this, we propose the Coherent Multimodal Reasoning Framework (CMRF), a novel approach that enhances LVLMs' common sense reasoning capabilities through an iterative, self-evaluating infer"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.02886","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.02886/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.02886","created_at":"2026-07-05T11:48:32.338438+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.02886v1","created_at":"2026-07-05T11:48:32.338438+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.02886","created_at":"2026-07-05T11:48:32.338438+00:00"},{"alias_kind":"pith_short_12","alias_value":"RESASOFFZHDI","created_at":"2026-07-05T11:48:32.338438+00:00"},{"alias_kind":"pith_short_16","alias_value":"RESASOFFZHDIKECM","created_at":"2026-07-05T11:48:32.338438+00:00"},{"alias_kind":"pith_short_8","alias_value":"RESASOFF","created_at":"2026-07-05T11:48:32.338438+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00079","citing_title":"Entropy-Guided Loop: Achieving Reasoning through Uncertainty-Aware Generation","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD","json":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD.json","graph_json":"https://pith.science/api/pith-number/RESASOFFZHDIKECM7KONTENQBD/graph.json","events_json":"https://pith.science/api/pith-number/RESASOFFZHDIKECM7KONTENQBD/events.json","paper":"https://pith.science/paper/RESASOFF"},"agent_actions":{"view_html":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD","download_json":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD.json","view_paper":"https://pith.science/paper/RESASOFF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.02886&json=true","fetch_graph":"https://pith.science/api/pith-number/RESASOFFZHDIKECM7KONTENQBD/graph.json","fetch_events":"https://pith.science/api/pith-number/RESASOFFZHDIKECM7KONTENQBD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD/action/storage_attestation","attest_author":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD/action/author_attestation","sign_citation":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD/action/citation_signature","submit_replication":"https://pith.science/pith/RESASOFFZHDIKECM7KONTENQBD/action/replication_record"}},"created_at":"2026-07-05T11:48:32.338438+00:00","updated_at":"2026-07-05T11:48:32.338438+00:00"}