{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QFAGXMYGS356EDMLUZRDFRHZYC","short_pith_number":"pith:QFAGXMYG","schema_version":"1.0","canonical_sha256":"81406bb30696fbe20d8ba66232c4f9c0ac99c9dd8024ea7da1d091bc1b18ae13","source":{"kind":"arxiv","id":"2402.04476","version":2},"attestation_state":"computed","paper":{"title":"Dual-View Visual Contextualization for Web Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Boyuan Zheng, Chan Hee Song, Jihyung Kil, Wei-Lun Chao, Xiang Deng, Yu Su","submitted_at":"2024-02-06T23:52:10Z","abstract_excerpt":"Automatic web navigation aims to build a web agent that can follow language instructions to execute complex and diverse tasks on real-world websites. Existing work primarily takes HTML documents as input, which define the contents and action spaces (i.e., actionable elements and operations) of webpages. Nevertheless, HTML documents may not provide a clear task-related context for each element, making it hard to select the right (sequence of) actions. In this paper, we propose to contextualize HTML elements through their \"dual views\" in webpage screenshots: each HTML element has its correspondi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.04476","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-06T23:52:10Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"857fb2d53839e0aace18ee345238fea941fdd2ecca780ad700e6b6d0f4d239d7","abstract_canon_sha256":"32d95fc52c178dbd820d92ffc756b498e75f58363b7869e35347b5409f7ca48d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:24.401799Z","signature_b64":"EvtxdltA4XpqryrBEID3YbK+Fevmu01nJZa+9HknNgUuX3DjU+erLr3xt3ySsTJbCCd9PTx/8WilCr/+KctGBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81406bb30696fbe20d8ba66232c4f9c0ac99c9dd8024ea7da1d091bc1b18ae13","last_reissued_at":"2026-07-05T08:02:24.401320Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:24.401320Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dual-View Visual Contextualization for Web Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Boyuan Zheng, Chan Hee Song, Jihyung Kil, Wei-Lun Chao, Xiang Deng, Yu Su","submitted_at":"2024-02-06T23:52:10Z","abstract_excerpt":"Automatic web navigation aims to build a web agent that can follow language instructions to execute complex and diverse tasks on real-world websites. Existing work primarily takes HTML documents as input, which define the contents and action spaces (i.e., actionable elements and operations) of webpages. Nevertheless, HTML documents may not provide a clear task-related context for each element, making it hard to select the right (sequence of) actions. In this paper, we propose to contextualize HTML elements through their \"dual views\" in webpage screenshots: each HTML element has its correspondi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04476","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04476/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.04476","created_at":"2026-07-05T08:02:24.401374+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.04476v2","created_at":"2026-07-05T08:02:24.401374+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04476","created_at":"2026-07-05T08:02:24.401374+00:00"},{"alias_kind":"pith_short_12","alias_value":"QFAGXMYGS356","created_at":"2026-07-05T08:02:24.401374+00:00"},{"alias_kind":"pith_short_16","alias_value":"QFAGXMYGS356EDML","created_at":"2026-07-05T08:02:24.401374+00:00"},{"alias_kind":"pith_short_8","alias_value":"QFAGXMYG","created_at":"2026-07-05T08:02:24.401374+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":283,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC","json":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC.json","graph_json":"https://pith.science/api/pith-number/QFAGXMYGS356EDMLUZRDFRHZYC/graph.json","events_json":"https://pith.science/api/pith-number/QFAGXMYGS356EDMLUZRDFRHZYC/events.json","paper":"https://pith.science/paper/QFAGXMYG"},"agent_actions":{"view_html":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC","download_json":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC.json","view_paper":"https://pith.science/paper/QFAGXMYG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.04476&json=true","fetch_graph":"https://pith.science/api/pith-number/QFAGXMYGS356EDMLUZRDFRHZYC/graph.json","fetch_events":"https://pith.science/api/pith-number/QFAGXMYGS356EDMLUZRDFRHZYC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC/action/storage_attestation","attest_author":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC/action/author_attestation","sign_citation":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC/action/citation_signature","submit_replication":"https://pith.science/pith/QFAGXMYGS356EDMLUZRDFRHZYC/action/replication_record"}},"created_at":"2026-07-05T08:02:24.401374+00:00","updated_at":"2026-07-05T08:02:24.401374+00:00"}