{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5OWTG55CQM26K45SG7FN5E74WE","short_pith_number":"pith:5OWTG55C","schema_version":"1.0","canonical_sha256":"ebad3377a28335e573b237cade93fcb1389921d080ba13bd563a618f50f83c51","source":{"kind":"arxiv","id":"2312.08914","version":3},"attestation_state":"computed","paper":{"title":"CogAgent: A Visual Language Model for GUI Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Xu, Jiazheng Xu, Jie Tang, Juanzi Li, Junhui Ji, Ming Ding, Qingsong Lv, Weihan Wang, Wenmeng Yu, Wenyi Hong, Yan Wang, Yuxiao Dong, Yuxuan Zhang, Zihan Wang","submitted_at":"2023-12-14T13:20:57Z","abstract_excerpt":"People are spending an enormous amount of time on digital devices through graphical user interfaces (GUIs), e.g., computer or smartphone screens. Large language models (LLMs) such as ChatGPT can assist people in tasks like writing emails, but struggle to understand and interact with GUIs, thus limiting their potential to increase automation levels. In this paper, we introduce CogAgent, an 18-billion-parameter visual language model (VLM) specializing in GUI understanding and navigation. By utilizing both low-resolution and high-resolution image encoders, CogAgent supports input at a resolution "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.08914","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-14T13:20:57Z","cross_cats_sorted":[],"title_canon_sha256":"aa5a0704f04c1339748622f898540f995331262d4d04648abebcfe9290bd7d90","abstract_canon_sha256":"9a10bdcfb541b92d0f0078da7155d4f09dce138f9ae9bbc2fd1eaa14f6329497"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:24.566994Z","signature_b64":"+c2pcZH6seLDTXkA/mObP0qf3x2TDuwJWcg6xXj8I5ZgPDuM9DgRgJk1DQ6vMbF25ihfpnbIekemQp9ov5cnCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ebad3377a28335e573b237cade93fcb1389921d080ba13bd563a618f50f83c51","last_reissued_at":"2026-07-05T09:54:24.566375Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:24.566375Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CogAgent: A Visual Language Model for GUI Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Xu, Jiazheng Xu, Jie Tang, Juanzi Li, Junhui Ji, Ming Ding, Qingsong Lv, Weihan Wang, Wenmeng Yu, Wenyi Hong, Yan Wang, Yuxiao Dong, Yuxuan Zhang, Zihan Wang","submitted_at":"2023-12-14T13:20:57Z","abstract_excerpt":"People are spending an enormous amount of time on digital devices through graphical user interfaces (GUIs), e.g., computer or smartphone screens. Large language models (LLMs) such as ChatGPT can assist people in tasks like writing emails, but struggle to understand and interact with GUIs, thus limiting their potential to increase automation levels. In this paper, we introduce CogAgent, an 18-billion-parameter visual language model (VLM) specializing in GUI understanding and navigation. By utilizing both low-resolution and high-resolution image encoders, CogAgent supports input at a resolution "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.08914","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.08914/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.08914","created_at":"2026-07-05T09:54:24.566454+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.08914v3","created_at":"2026-07-05T09:54:24.566454+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.08914","created_at":"2026-07-05T09:54:24.566454+00:00"},{"alias_kind":"pith_short_12","alias_value":"5OWTG55CQM26","created_at":"2026-07-05T09:54:24.566454+00:00"},{"alias_kind":"pith_short_16","alias_value":"5OWTG55CQM26K45S","created_at":"2026-07-05T09:54:24.566454+00:00"},{"alias_kind":"pith_short_8","alias_value":"5OWTG55C","created_at":"2026-07-05T09:54:24.566454+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20717","citing_title":"MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17321","citing_title":"ProCUA-SFT Technical Report","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13527","citing_title":"MMSkills: Towards Multimodal Skills for General Visual Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29445","citing_title":"Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29537","citing_title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29400","citing_title":"Architecture-Sensitive Supervised Fine-Tuning for Screen-Conditioned Action Prediction: A PiSAR Benchmark","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30884","citing_title":"GUI-C$^2$: Coarse-to-Fine GUI Grounding via Difficulty-Aware Reinforcement Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2406.08035","citing_title":"LVBench: An Extreme Long Video Understanding Benchmark","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04227","citing_title":"Mobile GUI Agents under Real-world Threats: Are We There Yet?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2401.10935","citing_title":"SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16994","citing_title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13527","citing_title":"MMSkills: Towards Multimodal Skills for General Visual Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13527","citing_title":"MMSkills: Towards Multimodal Skills for General Visual Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14573","citing_title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07972","citing_title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26622","citing_title":"OCR-Memory: Optical Context Retrieval for Long-Horizon Agent Memory","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08516","citing_title":"MolmoWeb: Open Visual Web Agent and Open Data for the Open Web","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14113","citing_title":"UI-Zoomer: Uncertainty-Driven Adaptive Zoom-In for GUI Grounding","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE","json":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE.json","graph_json":"https://pith.science/api/pith-number/5OWTG55CQM26K45SG7FN5E74WE/graph.json","events_json":"https://pith.science/api/pith-number/5OWTG55CQM26K45SG7FN5E74WE/events.json","paper":"https://pith.science/paper/5OWTG55C"},"agent_actions":{"view_html":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE","download_json":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE.json","view_paper":"https://pith.science/paper/5OWTG55C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.08914&json=true","fetch_graph":"https://pith.science/api/pith-number/5OWTG55CQM26K45SG7FN5E74WE/graph.json","fetch_events":"https://pith.science/api/pith-number/5OWTG55CQM26K45SG7FN5E74WE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE/action/storage_attestation","attest_author":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE/action/author_attestation","sign_citation":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE/action/citation_signature","submit_replication":"https://pith.science/pith/5OWTG55CQM26K45SG7FN5E74WE/action/replication_record"}},"created_at":"2026-07-05T09:54:24.566454+00:00","updated_at":"2026-07-05T09:54:24.566454+00:00"}