{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:42LJFWK4H6MHJVHAIQC6RZWMIY","short_pith_number":"pith:42LJFWK4","schema_version":"1.0","canonical_sha256":"e69692d95c3f9874d4e04405e8e6cc4635a0a30bcf41a6a760d1cdf5d430b95b","source":{"kind":"arxiv","id":"2505.21523","version":3},"attestation_state":"computed","paper":{"title":"More Thinking, Less Seeing? Assessing Amplified Hallucination in Multimodal Reasoning Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Chengzhi Liu, James Zou, Juncheng Wu, Qingyue Wei, Sheng Liu, Xin Eric Wang, Yuyin Zhou, Zhongxing Xu","submitted_at":"2025-05-23T05:08:40Z","abstract_excerpt":"Test-time compute has empowered multimodal large language models to generate extended reasoning chains, yielding strong performance on tasks such as multimodal math reasoning. However, this improved reasoning ability often comes with increased hallucination: as generations become longer, models tend to drift away from image-grounded content and rely more heavily on language priors. Attention analysis shows that longer reasoning chains lead to reduced focus on visual inputs, which contributes to hallucination. To systematically study this phenomenon, we introduce RH-AUC, a metric that quantifie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.21523","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T05:08:40Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"5d43c5c3968cfd708ecd52605f7c4b7249e9070c084cfd17fe4045d80abd8ae0","abstract_canon_sha256":"0866cb5100dafc0a517b853f4b06685c43b3e2e882b9abb769ebd3407f6b59e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:21.149370Z","signature_b64":"upgg62ubaHNvcZoAWJJMuwxD96bhv+3Gr/widEujyl3b+zL3IevWvVwZeb79Lgs2tt2Y5U/zG/MFs5DWOyRPAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e69692d95c3f9874d4e04405e8e6cc4635a0a30bcf41a6a760d1cdf5d430b95b","last_reissued_at":"2026-07-05T11:24:21.148849Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:21.148849Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"More Thinking, Less Seeing? Assessing Amplified Hallucination in Multimodal Reasoning Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Chengzhi Liu, James Zou, Juncheng Wu, Qingyue Wei, Sheng Liu, Xin Eric Wang, Yuyin Zhou, Zhongxing Xu","submitted_at":"2025-05-23T05:08:40Z","abstract_excerpt":"Test-time compute has empowered multimodal large language models to generate extended reasoning chains, yielding strong performance on tasks such as multimodal math reasoning. However, this improved reasoning ability often comes with increased hallucination: as generations become longer, models tend to drift away from image-grounded content and rely more heavily on language priors. Attention analysis shows that longer reasoning chains lead to reduced focus on visual inputs, which contributes to hallucination. To systematically study this phenomenon, we introduce RH-AUC, a metric that quantifie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.21523","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.21523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.21523","created_at":"2026-07-05T11:24:21.148911+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.21523v3","created_at":"2026-07-05T11:24:21.148911+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.21523","created_at":"2026-07-05T11:24:21.148911+00:00"},{"alias_kind":"pith_short_12","alias_value":"42LJFWK4H6MH","created_at":"2026-07-05T11:24:21.148911+00:00"},{"alias_kind":"pith_short_16","alias_value":"42LJFWK4H6MHJVHA","created_at":"2026-07-05T11:24:21.148911+00:00"},{"alias_kind":"pith_short_8","alias_value":"42LJFWK4","created_at":"2026-07-05T11:24:21.148911+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08059","citing_title":"When Thinking Hurts: Epistemic Signals in the Reasoning Chains of Visual Language Models","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19927","citing_title":"CARE: Competence-Aware Reward Shaping for Adaptive Reasoning Length in Video-MLLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04184","citing_title":"GroupToM-Bench: Benchmarking Group Theory of Mind and Nonlinear Social Emergence in MLLMs","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01558","citing_title":"Attention-guided Fine-tuning of Multimodal Large Language Models Improves Chain-of-Thought Reasoning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27195","citing_title":"EpiCurveBench: Evaluating VLMs on Epidemic Curve Digitization","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27298","citing_title":"Self-Ensembling Vision-Language Models for Chart Data Extraction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29496","citing_title":"On Asymmetric Optimization of Reasoning and Perception in Vision-Language Model Post-Training","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19192","citing_title":"Hallucination as Exploit: Evidence-Carrying Multimodal Agents","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14159","citing_title":"MVI-Bench: A Comprehensive Benchmark for Evaluating Robustness to Misleading Visual Inputs in LVLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14184","citing_title":"Deeper Thought, Weaker Aim: Understanding and Mitigating Perceptual Impairment during Reasoning in Multimodal Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18603","citing_title":"Starve to Perceive: Taming Lazy Perception in VLMs with Constrained Visual Bandwidth","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19192","citing_title":"Hallucination as Exploit: Evidence-Carrying Multimodal Agents","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19341","citing_title":"HalluWorld: A Controlled Benchmark for Hallucination via Reference World Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20177","citing_title":"From Seeing to Thinking: Decoupling Perception and Reasoning Improves Post-Training of Vision-Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19652","citing_title":"Self-Rewarding Vision-Language Model via Reasoning Decomposition","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08016","citing_title":"Video Parallel Scaling: Aggregating Diverse Frame Subsets for VideoLLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20490","citing_title":"RadAgents: Multimodal Agentic Reasoning for Chest X-ray Interpretation with Radiologist-like Workflows","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12623","citing_title":"Reasoning Within the Mind: Dynamic Multimodal Interleaving in Latent Space","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17555","citing_title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03179","citing_title":"Understanding the Role of Hallucination in Reinforcement Post-Training of Multimodal Reasoning Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09266","citing_title":"SeePhys Pro: Diagnosing Modality Transfer and Blind-Training Effects in Multimodal RLVR for Physics Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11931","citing_title":"Learn to Think: Improving Multimodal Reasoning through Vision-Aware Self-Improvement Training","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09266","citing_title":"SeePhys Pro: Diagnosing Modality Transfer and Blind-Training Effects in Multimodal RLVR for Physics Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24583","citing_title":"Improving Vision-language Models with Perception-centric Process Reward Models","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY","json":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY.json","graph_json":"https://pith.science/api/pith-number/42LJFWK4H6MHJVHAIQC6RZWMIY/graph.json","events_json":"https://pith.science/api/pith-number/42LJFWK4H6MHJVHAIQC6RZWMIY/events.json","paper":"https://pith.science/paper/42LJFWK4"},"agent_actions":{"view_html":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY","download_json":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY.json","view_paper":"https://pith.science/paper/42LJFWK4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.21523&json=true","fetch_graph":"https://pith.science/api/pith-number/42LJFWK4H6MHJVHAIQC6RZWMIY/graph.json","fetch_events":"https://pith.science/api/pith-number/42LJFWK4H6MHJVHAIQC6RZWMIY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY/action/storage_attestation","attest_author":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY/action/author_attestation","sign_citation":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY/action/citation_signature","submit_replication":"https://pith.science/pith/42LJFWK4H6MHJVHAIQC6RZWMIY/action/replication_record"}},"created_at":"2026-07-05T11:24:21.148911+00:00","updated_at":"2026-07-05T11:24:21.148911+00:00"}