{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FQH4YXYCEQ2XRXBUAV64SUYZMZ","short_pith_number":"pith:FQH4YXYC","schema_version":"1.0","canonical_sha256":"2c0fcc5f02243578dc34057dc95319665a2a42f8ecfdc2c5a039f43efcb22358","source":{"kind":"arxiv","id":"2410.19100","version":3},"attestation_state":"computed","paper":{"title":"VideoWebArena: Evaluating Long Context Multimodal Agents with Video Understanding Web Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Charles Ding, Dan Zhao, Justin Lin, Kazuhito Koishida, Lawrence Jang, Paul Pu Liang, Rogerio Bonatti, Yinheng Li","submitted_at":"2024-10-24T19:03:01Z","abstract_excerpt":"Videos are often used to learn or extract the necessary information to complete tasks in ways different than what text and static imagery alone can provide. However, many existing agent benchmarks neglect long-context video understanding, instead focusing on text or static image inputs. To bridge this gap, we introduce VideoWebArena (VideoWA), a benchmark for evaluating the capabilities of long-context multimodal agents for video understanding. VideoWA consists of 2,021 web agent tasks based on manually crafted video tutorials, which total almost four hours of content. For our benchmark, we de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.19100","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-24T19:03:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3d72fedb3b6e9c537d960b41263a530f84f5d678539c0b2756bfac2b764a8e51","abstract_canon_sha256":"caf2005ebd3af02c85bb8d5eca2bbc1639c00f988a38d8f5938e719f9495669b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:47.121356Z","signature_b64":"O+2CIX/ZYPr9063JNF9xXGA6wY7l85VMoSvrF93nFPzm1mt2RDv0RDUzf3Hibz3xdidNfd6tt7jkR+UReOOXDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c0fcc5f02243578dc34057dc95319665a2a42f8ecfdc2c5a039f43efcb22358","last_reissued_at":"2026-07-05T10:14:47.120826Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:47.120826Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoWebArena: Evaluating Long Context Multimodal Agents with Video Understanding Web Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Charles Ding, Dan Zhao, Justin Lin, Kazuhito Koishida, Lawrence Jang, Paul Pu Liang, Rogerio Bonatti, Yinheng Li","submitted_at":"2024-10-24T19:03:01Z","abstract_excerpt":"Videos are often used to learn or extract the necessary information to complete tasks in ways different than what text and static imagery alone can provide. However, many existing agent benchmarks neglect long-context video understanding, instead focusing on text or static image inputs. To bridge this gap, we introduce VideoWebArena (VideoWA), a benchmark for evaluating the capabilities of long-context multimodal agents for video understanding. VideoWA consists of 2,021 web agent tasks based on manually crafted video tutorials, which total almost four hours of content. For our benchmark, we de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19100","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.19100/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.19100","created_at":"2026-07-05T10:14:47.120885+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.19100v3","created_at":"2026-07-05T10:14:47.120885+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19100","created_at":"2026-07-05T10:14:47.120885+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQH4YXYCEQ2X","created_at":"2026-07-05T10:14:47.120885+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQH4YXYCEQ2XRXBU","created_at":"2026-07-05T10:14:47.120885+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQH4YXYC","created_at":"2026-07-05T10:14:47.120885+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03951","citing_title":"Demo2Tutorial: From Human Experience to Multimodal Software Tutorials","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29472","citing_title":"Agent-Computer Observation Interfaces Enable Dynamic Computer Use","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.06943","citing_title":"Watching, Reasoning, and Searching: A Video Deep Research Benchmark on Open Web for Agentic Video Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18758","citing_title":"OmniGUI: Benchmarking GUI Agents in Omni-Modal Smartphone Environments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19662","citing_title":"FieldWorkArena: Agentic AI Benchmark for Real Field Work Tasks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.12538","citing_title":"Agentic Reasoning for Large Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12703","citing_title":"MMCL-Bench: Multimodal Context Learning from Visual Rules, Procedures, and Evidence","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10966","citing_title":"MMTB: Evaluating Terminal Agents on Multimedia-File Tasks","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ","json":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ.json","graph_json":"https://pith.science/api/pith-number/FQH4YXYCEQ2XRXBUAV64SUYZMZ/graph.json","events_json":"https://pith.science/api/pith-number/FQH4YXYCEQ2XRXBUAV64SUYZMZ/events.json","paper":"https://pith.science/paper/FQH4YXYC"},"agent_actions":{"view_html":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ","download_json":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ.json","view_paper":"https://pith.science/paper/FQH4YXYC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.19100&json=true","fetch_graph":"https://pith.science/api/pith-number/FQH4YXYCEQ2XRXBUAV64SUYZMZ/graph.json","fetch_events":"https://pith.science/api/pith-number/FQH4YXYCEQ2XRXBUAV64SUYZMZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ/action/storage_attestation","attest_author":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ/action/author_attestation","sign_citation":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ/action/citation_signature","submit_replication":"https://pith.science/pith/FQH4YXYCEQ2XRXBUAV64SUYZMZ/action/replication_record"}},"created_at":"2026-07-05T10:14:47.120885+00:00","updated_at":"2026-07-05T10:14:47.120885+00:00"}