{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CL3OXFOQ5NDWQOR53RFCLTRCIA","short_pith_number":"pith:CL3OXFOQ","schema_version":"1.0","canonical_sha256":"12f6eb95d0eb47683a3ddc4a25ce22403430f0f3a5f90dc6b4d1c81233f24fba","source":{"kind":"arxiv","id":"2508.13186","version":1},"attestation_state":"computed","paper":{"title":"MM-BrowseComp: A Comprehensive Benchmark for Multimodal Browsing Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Chenchen Jing, Chenchen Zhang, Chuanhao Li, Ge Zhang, Hao Lu, Haoyang He, Haozhe Zhang, Jiaheng Liu, Jian Yang, Jiayi Tian, Jihao Gu, Jun Dong, Ruizhe Ding, Shilei Wen, Shilong Li, Tianhao Peng, Wangchunshu Zhou, Wenhao Huang, Wenjie Wang, Xingyuan Bu, Yancheng He, Yuanxing Zhang, Zhaoxiang Zhang, Zhen Li","submitted_at":"2025-08-14T13:46:47Z","abstract_excerpt":"AI agents with advanced reasoning and tool use capabilities have demonstrated impressive performance in web browsing for deep search. While existing benchmarks such as BrowseComp evaluate these browsing abilities, they primarily focus on textual information, overlooking the prevalence of multimodal content. To bridge this gap, we introduce MM-BrowseComp, a novel benchmark comprising 224 challenging, hand-crafted questions specifically designed to assess agents' multimodal retrieval and reasoning capabilities. These questions often incorporate images in prompts, and crucial information encounte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.13186","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-14T13:46:47Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"6235548834894956d4b018533801c9f471e7285f8103df1f9010bdce81462091","abstract_canon_sha256":"a3186868d77563a983c54a07c84dc9fc8bcefc51bd60099bff6bcc2a620ce11c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:43.976612Z","signature_b64":"OAfNSigeqWYACywC4uY4k3szYvBFMX8WWLbYwf+QZeOduY4yWuxHJOgblIgroQukNoC7qIQt17Box+j8yzADDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"12f6eb95d0eb47683a3ddc4a25ce22403430f0f3a5f90dc6b4d1c81233f24fba","last_reissued_at":"2026-07-05T11:55:43.976108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:43.976108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MM-BrowseComp: A Comprehensive Benchmark for Multimodal Browsing Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Chenchen Jing, Chenchen Zhang, Chuanhao Li, Ge Zhang, Hao Lu, Haoyang He, Haozhe Zhang, Jiaheng Liu, Jian Yang, Jiayi Tian, Jihao Gu, Jun Dong, Ruizhe Ding, Shilei Wen, Shilong Li, Tianhao Peng, Wangchunshu Zhou, Wenhao Huang, Wenjie Wang, Xingyuan Bu, Yancheng He, Yuanxing Zhang, Zhaoxiang Zhang, Zhen Li","submitted_at":"2025-08-14T13:46:47Z","abstract_excerpt":"AI agents with advanced reasoning and tool use capabilities have demonstrated impressive performance in web browsing for deep search. While existing benchmarks such as BrowseComp evaluate these browsing abilities, they primarily focus on textual information, overlooking the prevalence of multimodal content. To bridge this gap, we introduce MM-BrowseComp, a novel benchmark comprising 224 challenging, hand-crafted questions specifically designed to assess agents' multimodal retrieval and reasoning capabilities. These questions often incorporate images in prompts, and crucial information encounte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.13186","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.13186/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.13186","created_at":"2026-07-05T11:55:43.976169+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.13186v1","created_at":"2026-07-05T11:55:43.976169+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.13186","created_at":"2026-07-05T11:55:43.976169+00:00"},{"alias_kind":"pith_short_12","alias_value":"CL3OXFOQ5NDW","created_at":"2026-07-05T11:55:43.976169+00:00"},{"alias_kind":"pith_short_16","alias_value":"CL3OXFOQ5NDWQOR5","created_at":"2026-07-05T11:55:43.976169+00:00"},{"alias_kind":"pith_short_8","alias_value":"CL3OXFOQ","created_at":"2026-07-05T11:55:43.976169+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10460","citing_title":"LakeQA: An Exploratory QA Benchmark over a Million-Scale Data Lake","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00248","citing_title":"Seed2.0 Model Card: Towards Intelligence Frontier for Real-World Complexity","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07689","citing_title":"Struct-Searcher: Agentic Structural Thinking Advances Multimodal Deep Information Seeking","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10832","citing_title":"Towards On-Policy Data Evolution for Visual-Native Multimodal Deep Search Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17946","citing_title":"SVFSearch: A Multimodal Knowledge-Intensive Benchmark for Short-Video Frame Search in the Gaming Vertical Domain","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17946","citing_title":"SVFSearch: A Multimodal Knowledge-Intensive Benchmark for Short-Video Frame Search in the Gaming Vertical Domain","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20633","citing_title":"Seed1.8 Model Card: Towards Generalized Real-World Agency","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04017","citing_title":"GeoBrowse: A Geolocation Benchmark for Agentic Tool Use with Expert-Annotated Reasoning Traces","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12481","citing_title":"ToolCUA: Towards Optimal GUI-Tool Path Orchestration for Computer Use Agents","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10832","citing_title":"Towards On-Policy Data Evolution for Visual-Native Multimodal Deep Search Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12890","citing_title":"Towards Long-horizon Agentic Multimodal Search","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14448","citing_title":"MARCA: A Checklist-Based Benchmark for Multilingual Web Search","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA","json":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA.json","graph_json":"https://pith.science/api/pith-number/CL3OXFOQ5NDWQOR53RFCLTRCIA/graph.json","events_json":"https://pith.science/api/pith-number/CL3OXFOQ5NDWQOR53RFCLTRCIA/events.json","paper":"https://pith.science/paper/CL3OXFOQ"},"agent_actions":{"view_html":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA","download_json":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA.json","view_paper":"https://pith.science/paper/CL3OXFOQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.13186&json=true","fetch_graph":"https://pith.science/api/pith-number/CL3OXFOQ5NDWQOR53RFCLTRCIA/graph.json","fetch_events":"https://pith.science/api/pith-number/CL3OXFOQ5NDWQOR53RFCLTRCIA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA/action/storage_attestation","attest_author":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA/action/author_attestation","sign_citation":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA/action/citation_signature","submit_replication":"https://pith.science/pith/CL3OXFOQ5NDWQOR53RFCLTRCIA/action/replication_record"}},"created_at":"2026-07-05T11:55:43.976169+00:00","updated_at":"2026-07-05T11:55:43.976169+00:00"}