{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TN2ZY6LRKYML2DTGY7XIKBCTQO","short_pith_number":"pith:TN2ZY6LR","schema_version":"1.0","canonical_sha256":"9b759c79715618bd0e66c7ee85045383a3edf5c2ed1287ab8afbce0816fa8535","source":{"kind":"arxiv","id":"2405.18358","version":1},"attestation_state":"computed","paper":{"title":"MMCTAgent: Multi-modal Critical Thinking Agent Framework for Complex Visual Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Akshay Nambi, Somnath Kumar, Tanuja Ganu, Yash Gadhia","submitted_at":"2024-05-28T16:55:41Z","abstract_excerpt":"Recent advancements in Multi-modal Large Language Models (MLLMs) have significantly improved their performance in tasks combining vision and language. However, challenges persist in detailed multi-modal understanding, comprehension of complex tasks, and reasoning over multi-modal information. This paper introduces MMCTAgent, a novel multi-modal critical thinking agent framework designed to address the inherent limitations of current MLLMs in complex visual reasoning tasks. Inspired by human cognitive processes and critical thinking, MMCTAgent iteratively analyzes multi-modal information, decom"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.18358","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-28T16:55:41Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"6276e882a1ffba7ab45cdbd636b3db7e09dee1cf51bf707e10b52040795cd4a4","abstract_canon_sha256":"4df8b0d69dd12e4c5274cc080d85e2bdda9d10cc1bbcc7ec5ccd232dcb805f5d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:19.038884Z","signature_b64":"ITs7eD37plPWYtyZcQYAC0o5ySy0OhknfqZuS4GpVd2FK44R7NwrerDhNKcofqN8FPa5bRm0JKQhcW/pLdG5Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b759c79715618bd0e66c7ee85045383a3edf5c2ed1287ab8afbce0816fa8535","last_reissued_at":"2026-07-05T08:24:19.038465Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:19.038465Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMCTAgent: Multi-modal Critical Thinking Agent Framework for Complex Visual Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Akshay Nambi, Somnath Kumar, Tanuja Ganu, Yash Gadhia","submitted_at":"2024-05-28T16:55:41Z","abstract_excerpt":"Recent advancements in Multi-modal Large Language Models (MLLMs) have significantly improved their performance in tasks combining vision and language. However, challenges persist in detailed multi-modal understanding, comprehension of complex tasks, and reasoning over multi-modal information. This paper introduces MMCTAgent, a novel multi-modal critical thinking agent framework designed to address the inherent limitations of current MLLMs in complex visual reasoning tasks. Inspired by human cognitive processes and critical thinking, MMCTAgent iteratively analyzes multi-modal information, decom"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.18358","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.18358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.18358","created_at":"2026-07-05T08:24:19.038522+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.18358v1","created_at":"2026-07-05T08:24:19.038522+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.18358","created_at":"2026-07-05T08:24:19.038522+00:00"},{"alias_kind":"pith_short_12","alias_value":"TN2ZY6LRKYML","created_at":"2026-07-05T08:24:19.038522+00:00"},{"alias_kind":"pith_short_16","alias_value":"TN2ZY6LRKYML2DTG","created_at":"2026-07-05T08:24:19.038522+00:00"},{"alias_kind":"pith_short_8","alias_value":"TN2ZY6LR","created_at":"2026-07-05T08:24:19.038522+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08497","citing_title":"Cognitive-structured Multimodal Agent for Multimodal Understanding, Generation, and Editing","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2508.10016","citing_title":"Training-Free Multimodal Large Language Model Orchestration","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10016","citing_title":"Training-Free Multimodal Large Language Model Orchestration","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03438","citing_title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO","json":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO.json","graph_json":"https://pith.science/api/pith-number/TN2ZY6LRKYML2DTGY7XIKBCTQO/graph.json","events_json":"https://pith.science/api/pith-number/TN2ZY6LRKYML2DTGY7XIKBCTQO/events.json","paper":"https://pith.science/paper/TN2ZY6LR"},"agent_actions":{"view_html":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO","download_json":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO.json","view_paper":"https://pith.science/paper/TN2ZY6LR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.18358&json=true","fetch_graph":"https://pith.science/api/pith-number/TN2ZY6LRKYML2DTGY7XIKBCTQO/graph.json","fetch_events":"https://pith.science/api/pith-number/TN2ZY6LRKYML2DTGY7XIKBCTQO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO/action/storage_attestation","attest_author":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO/action/author_attestation","sign_citation":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO/action/citation_signature","submit_replication":"https://pith.science/pith/TN2ZY6LRKYML2DTGY7XIKBCTQO/action/replication_record"}},"created_at":"2026-07-05T08:24:19.038522+00:00","updated_at":"2026-07-05T08:24:19.038522+00:00"}