{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RKPT3BUFX53UZ57GUCLLQWNYVS","short_pith_number":"pith:RKPT3BUF","schema_version":"1.0","canonical_sha256":"8a9f3d8685bf774cf7e6a096b859b8acb80befc5696c5c858cfe3c236e177321","source":{"kind":"arxiv","id":"2407.01511","version":4},"attestation_state":"computed","paper":{"title":"CRAB: Cross-environment Agent Benchmark for Multimodal Language Model Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Anjie Yang, Bernard Ghanem, Bochen Qian, Dai-Jie Wu, Guohao Li, Jianbo Deng, Linyao Chen, Philip Torr, Shilong Liu, Tianqi Xu, Xiang Yao, Yanjun Chen, Yongchao Chen, Zecheng Zhang, Zhaoxuan Jin, Zhiqiang Xie","submitted_at":"2024-07-01T17:55:04Z","abstract_excerpt":"The development of autonomous agents increasingly relies on Multimodal Language Models (MLMs) to perform tasks described in natural language with GUI environments, such as websites, desktop computers, or mobile phones. Existing benchmarks for MLM agents in interactive environments are limited by their focus on a single environment, lack of detailed and generalized evaluation methods, and the complexities of constructing tasks and evaluators. To overcome these limitations, we introduce Crab, the first agent benchmark framework designed to support cross-environment tasks, incorporating a graph-b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.01511","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-07-01T17:55:04Z","cross_cats_sorted":[],"title_canon_sha256":"b380cf2637a6aa1991ab41997d62efbb46c7502e7a1b8dc7f3aa6ec7c90a467d","abstract_canon_sha256":"53e2907fc78990d095de632a80489704372e701e92527f4a22434f144f95adff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:39:41.427121Z","signature_b64":"7BI8lWVMM42clRqwt5sJ4l/gm5HjDwIxGB37Hde1xxlDd2dhtmy2x8hAXPFL8d8rJFtAh/ktCTyNHLnkra39Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a9f3d8685bf774cf7e6a096b859b8acb80befc5696c5c858cfe3c236e177321","last_reissued_at":"2026-07-05T11:39:41.426582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:39:41.426582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CRAB: Cross-environment Agent Benchmark for Multimodal Language Model Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Anjie Yang, Bernard Ghanem, Bochen Qian, Dai-Jie Wu, Guohao Li, Jianbo Deng, Linyao Chen, Philip Torr, Shilong Liu, Tianqi Xu, Xiang Yao, Yanjun Chen, Yongchao Chen, Zecheng Zhang, Zhaoxuan Jin, Zhiqiang Xie","submitted_at":"2024-07-01T17:55:04Z","abstract_excerpt":"The development of autonomous agents increasingly relies on Multimodal Language Models (MLMs) to perform tasks described in natural language with GUI environments, such as websites, desktop computers, or mobile phones. Existing benchmarks for MLM agents in interactive environments are limited by their focus on a single environment, lack of detailed and generalized evaluation methods, and the complexities of constructing tasks and evaluators. To overcome these limitations, we introduce Crab, the first agent benchmark framework designed to support cross-environment tasks, incorporating a graph-b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.01511","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.01511/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.01511","created_at":"2026-07-05T11:39:41.426652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.01511v4","created_at":"2026-07-05T11:39:41.426652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.01511","created_at":"2026-07-05T11:39:41.426652+00:00"},{"alias_kind":"pith_short_12","alias_value":"RKPT3BUFX53U","created_at":"2026-07-05T11:39:41.426652+00:00"},{"alias_kind":"pith_short_16","alias_value":"RKPT3BUFX53UZ57G","created_at":"2026-07-05T11:39:41.426652+00:00"},{"alias_kind":"pith_short_8","alias_value":"RKPT3BUF","created_at":"2026-07-05T11:39:41.426652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27761","citing_title":"AndroidDaily: A Verifiable Benchmark for Mobile GUI Agents on Real-World Closed-Source Applications","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14201","citing_title":"ExCyTIn-Bench: Evaluating LLM agents on Cyber Threat Investigation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04454","citing_title":"Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2508.07407","citing_title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13564","citing_title":"Memory in the Age of AI Agents","ref_index":289,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS","json":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS.json","graph_json":"https://pith.science/api/pith-number/RKPT3BUFX53UZ57GUCLLQWNYVS/graph.json","events_json":"https://pith.science/api/pith-number/RKPT3BUFX53UZ57GUCLLQWNYVS/events.json","paper":"https://pith.science/paper/RKPT3BUF"},"agent_actions":{"view_html":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS","download_json":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS.json","view_paper":"https://pith.science/paper/RKPT3BUF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.01511&json=true","fetch_graph":"https://pith.science/api/pith-number/RKPT3BUFX53UZ57GUCLLQWNYVS/graph.json","fetch_events":"https://pith.science/api/pith-number/RKPT3BUFX53UZ57GUCLLQWNYVS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS/action/storage_attestation","attest_author":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS/action/author_attestation","sign_citation":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS/action/citation_signature","submit_replication":"https://pith.science/pith/RKPT3BUFX53UZ57GUCLLQWNYVS/action/replication_record"}},"created_at":"2026-07-05T11:39:41.426652+00:00","updated_at":"2026-07-05T11:39:41.426652+00:00"}