{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OYNPE5HKDRCQ7DNJMYGJDNAA42","short_pith_number":"pith:OYNPE5HK","schema_version":"1.0","canonical_sha256":"761af274ea1c450f8da9660c91b400e6871b8df7cfa4ed4c1d32521104b3d125","source":{"kind":"arxiv","id":"2402.15116","version":1},"attestation_state":"computed","paper":{"title":"Large Multimodal Agents: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Guanbin Li, Junlin Xie, Ruifei Zhang, Xiang Wan, Zhihong Chen","submitted_at":"2024-02-23T06:04:23Z","abstract_excerpt":"Large language models (LLMs) have achieved superior performance in powering text-based AI agents, endowing them with decision-making and reasoning abilities akin to humans. Concurrently, there is an emerging research trend focused on extending these LLM-powered AI agents into the multimodal domain. This extension enables AI agents to interpret and respond to diverse multimodal user queries, thereby handling more intricate and nuanced tasks. In this paper, we conduct a systematic review of LLM-driven multimodal agents, which we refer to as large multimodal agents ( LMAs for short). First, we in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.15116","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-23T06:04:23Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"290e7c5d80007c0e5d46cd6f3a92278d247659e335fd272448e4cd7cb814630f","abstract_canon_sha256":"ebf4dc8ac0c8182546d4a75a3b645e3aa990f9bffdb3d0c27c99da40f88cc109"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:36.134692Z","signature_b64":"sYwPN2fQzmf+GtZAp7jFp7OZ07IaOL04wPS05G3Mixi1KrebHqDMik+U/uk0NLOcdrfn/lglPiMylc8b/e3VBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"761af274ea1c450f8da9660c91b400e6871b8df7cfa4ed4c1d32521104b3d125","last_reissued_at":"2026-07-05T07:48:36.134230Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:36.134230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Multimodal Agents: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Guanbin Li, Junlin Xie, Ruifei Zhang, Xiang Wan, Zhihong Chen","submitted_at":"2024-02-23T06:04:23Z","abstract_excerpt":"Large language models (LLMs) have achieved superior performance in powering text-based AI agents, endowing them with decision-making and reasoning abilities akin to humans. Concurrently, there is an emerging research trend focused on extending these LLM-powered AI agents into the multimodal domain. This extension enables AI agents to interpret and respond to diverse multimodal user queries, thereby handling more intricate and nuanced tasks. In this paper, we conduct a systematic review of LLM-driven multimodal agents, which we refer to as large multimodal agents ( LMAs for short). First, we in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.15116","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.15116/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.15116","created_at":"2026-07-05T07:48:36.134288+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.15116v1","created_at":"2026-07-05T07:48:36.134288+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.15116","created_at":"2026-07-05T07:48:36.134288+00:00"},{"alias_kind":"pith_short_12","alias_value":"OYNPE5HKDRCQ","created_at":"2026-07-05T07:48:36.134288+00:00"},{"alias_kind":"pith_short_16","alias_value":"OYNPE5HKDRCQ7DNJ","created_at":"2026-07-05T07:48:36.134288+00:00"},{"alias_kind":"pith_short_8","alias_value":"OYNPE5HK","created_at":"2026-07-05T07:48:36.134288+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11702","citing_title":"MedCTA: A Benchmark for Clinical Tool Agents","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09669","citing_title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15128","citing_title":"MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2407.13193","citing_title":"Retrieval-Augmented Generation for Natural Language Processing: A Survey","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16120","citing_title":"LLM-Powered AI Agent Systems and Their Applications in Industry","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19662","citing_title":"FieldWorkArena: Agentic AI Benchmark for Real Field Work Tasks","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14133","citing_title":"Formalizing the Safety, Security, and Functional Properties of Agentic AI Systems","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2407.01284","citing_title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08392","citing_title":"ST-BiBench: Benchmarking Multi-Stream Multimodal Coordination in Bimanual Embodied Tasks for MLLMs","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2602.22683","citing_title":"SUPERGLASSES: Benchmarking Vision Language Models as Intelligent Agents for AI Smart Glasses","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01455","citing_title":"From Verbatim to Gist: Distilling Pyramidal Multimodal Memory via Semantic Information Bottleneck for Long-Horizon Video Agents","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":298,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13213","citing_title":"Hierarchical Attacks for Multi-Modal Multi-Agent Reasoning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17052","citing_title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42","json":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42.json","graph_json":"https://pith.science/api/pith-number/OYNPE5HKDRCQ7DNJMYGJDNAA42/graph.json","events_json":"https://pith.science/api/pith-number/OYNPE5HKDRCQ7DNJMYGJDNAA42/events.json","paper":"https://pith.science/paper/OYNPE5HK"},"agent_actions":{"view_html":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42","download_json":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42.json","view_paper":"https://pith.science/paper/OYNPE5HK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.15116&json=true","fetch_graph":"https://pith.science/api/pith-number/OYNPE5HKDRCQ7DNJMYGJDNAA42/graph.json","fetch_events":"https://pith.science/api/pith-number/OYNPE5HKDRCQ7DNJMYGJDNAA42/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42/action/storage_attestation","attest_author":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42/action/author_attestation","sign_citation":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42/action/citation_signature","submit_replication":"https://pith.science/pith/OYNPE5HKDRCQ7DNJMYGJDNAA42/action/replication_record"}},"created_at":"2026-07-05T07:48:36.134288+00:00","updated_at":"2026-07-05T07:48:36.134288+00:00"}