{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ETEZNWJISIQYIVV52EKG7QNBBQ","short_pith_number":"pith:ETEZNWJI","schema_version":"1.0","canonical_sha256":"24c996d92892218456bdd1146fc1a10c2c1a1b18f153943491fca37d39c1ffda","source":{"kind":"arxiv","id":"2403.19322","version":2},"attestation_state":"computed","paper":{"title":"Plug-and-Play Grounding of Reasoning in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dehu Li, Jiaxing Chen, Weimo Deng, Xiang An, Yin Xie, Yongle Zhao, Yuxuan Liu, Ziyong Feng","submitted_at":"2024-03-28T11:26:30Z","abstract_excerpt":"The rise of Multimodal Large Language Models (MLLMs), renowned for their advanced instruction-following and reasoning capabilities, has significantly propelled the field of visual reasoning. However, due to limitations in their image tokenization processes, most MLLMs struggle to capture fine details of text and objects in images, especially in high-resolution samples. To overcome this limitation, we introduce P2G, a novel framework for plug-and-play grounding in MLLMs. P2G utilizes the tool-usage potential of MLLMs to employ expert agents for on-the-fly grounding of reasoning into critical vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.19322","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-28T11:26:30Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5ca098a95d3e006b836bb1ad061a7bb2d7c914563058fe4ebd4ba9dcdb305eec","abstract_canon_sha256":"23e815ddaba41692632bc434b790e269f9f4a5cbdfff4c6401f58cc6211815ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:38.241820Z","signature_b64":"3r3ZXiMaYCxnFWRjIEPmfxOpOHmh431pCWcY9O7SNZJrfwdMV23auJYsyM9d7NEW6XjoM8wrAToHmcw1ya7RAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24c996d92892218456bdd1146fc1a10c2c1a1b18f153943491fca37d39c1ffda","last_reissued_at":"2026-07-05T08:33:38.241336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:38.241336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Plug-and-Play Grounding of Reasoning in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dehu Li, Jiaxing Chen, Weimo Deng, Xiang An, Yin Xie, Yongle Zhao, Yuxuan Liu, Ziyong Feng","submitted_at":"2024-03-28T11:26:30Z","abstract_excerpt":"The rise of Multimodal Large Language Models (MLLMs), renowned for their advanced instruction-following and reasoning capabilities, has significantly propelled the field of visual reasoning. However, due to limitations in their image tokenization processes, most MLLMs struggle to capture fine details of text and objects in images, especially in high-resolution samples. To overcome this limitation, we introduce P2G, a novel framework for plug-and-play grounding in MLLMs. P2G utilizes the tool-usage potential of MLLMs to employ expert agents for on-the-fly grounding of reasoning into critical vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19322","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.19322/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.19322","created_at":"2026-07-05T08:33:38.241393+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.19322v2","created_at":"2026-07-05T08:33:38.241393+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19322","created_at":"2026-07-05T08:33:38.241393+00:00"},{"alias_kind":"pith_short_12","alias_value":"ETEZNWJISIQY","created_at":"2026-07-05T08:33:38.241393+00:00"},{"alias_kind":"pith_short_16","alias_value":"ETEZNWJISIQYIVV5","created_at":"2026-07-05T08:33:38.241393+00:00"},{"alias_kind":"pith_short_8","alias_value":"ETEZNWJI","created_at":"2026-07-05T08:33:38.241393+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31174","citing_title":"Detect in Any Scene: An Agentic Framework for Object Detection with Experience-Aware Reasoning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00562","citing_title":"DeepLatent: Think with Images via Parallel Latent Visual Reasoning","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ","json":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ.json","graph_json":"https://pith.science/api/pith-number/ETEZNWJISIQYIVV52EKG7QNBBQ/graph.json","events_json":"https://pith.science/api/pith-number/ETEZNWJISIQYIVV52EKG7QNBBQ/events.json","paper":"https://pith.science/paper/ETEZNWJI"},"agent_actions":{"view_html":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ","download_json":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ.json","view_paper":"https://pith.science/paper/ETEZNWJI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.19322&json=true","fetch_graph":"https://pith.science/api/pith-number/ETEZNWJISIQYIVV52EKG7QNBBQ/graph.json","fetch_events":"https://pith.science/api/pith-number/ETEZNWJISIQYIVV52EKG7QNBBQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ/action/storage_attestation","attest_author":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ/action/author_attestation","sign_citation":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ/action/citation_signature","submit_replication":"https://pith.science/pith/ETEZNWJISIQYIVV52EKG7QNBBQ/action/replication_record"}},"created_at":"2026-07-05T08:33:38.241393+00:00","updated_at":"2026-07-05T08:33:38.241393+00:00"}