{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BEYIZR4EWTBG3OF5DV5UPYE46Z","short_pith_number":"pith:BEYIZR4E","schema_version":"1.0","canonical_sha256":"09308cc784b4c26db8bd1d7b47e09cf646431dbf42e299d3ac0c14d5ecd1432f","source":{"kind":"arxiv","id":"2410.19461","version":2},"attestation_state":"computed","paper":{"title":"EDGE: Enhanced Grounded GUI Understanding with Enriched Multi-Granularity Synthetic Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Deqing Yang, Hangcheng Li, Jiaqing Liang, Sihang Jiang, Xuetian Chen","submitted_at":"2024-10-25T10:46:17Z","abstract_excerpt":"Autonomous agents operating on the graphical user interfaces (GUIs) of various applications hold immense practical value. Unlike the large language model (LLM)-based methods which rely on structured texts and customized backends, the approaches using large vision-language models (LVLMs) are more intuitive and adaptable as they can visually perceive and directly interact with screens, making them indispensable in general scenarios without text metadata and tailored backends. Given the lack of high-quality training data for GUI-related tasks in existing work, this paper aims to enhance the GUI u"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.19461","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-25T10:46:17Z","cross_cats_sorted":[],"title_canon_sha256":"4ae25dda7b83659e79d84111f8093fbf4f6b340ba3558371109fd6ec3fdad886","abstract_canon_sha256":"d38784c7f85a21b89bb26f6e3ece139e80114ba9107532f54adc7bd9e2e793ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:23.304034Z","signature_b64":"o7KudRMgq4PY0+z2p+Qe5qFHW+++0NiRUZVJkIwX8Fv9djRrPJenwZaLdjvlebLWMNAXqCq/ZLuFAg2DlKKOCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"09308cc784b4c26db8bd1d7b47e09cf646431dbf42e299d3ac0c14d5ecd1432f","last_reissued_at":"2026-07-05T09:30:23.303580Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:23.303580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EDGE: Enhanced Grounded GUI Understanding with Enriched Multi-Granularity Synthetic Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Deqing Yang, Hangcheng Li, Jiaqing Liang, Sihang Jiang, Xuetian Chen","submitted_at":"2024-10-25T10:46:17Z","abstract_excerpt":"Autonomous agents operating on the graphical user interfaces (GUIs) of various applications hold immense practical value. Unlike the large language model (LLM)-based methods which rely on structured texts and customized backends, the approaches using large vision-language models (LVLMs) are more intuitive and adaptable as they can visually perceive and directly interact with screens, making them indispensable in general scenarios without text metadata and tailored backends. Given the lack of high-quality training data for GUI-related tasks in existing work, this paper aims to enhance the GUI u"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19461","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.19461/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.19461","created_at":"2026-07-05T09:30:23.303648+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.19461v2","created_at":"2026-07-05T09:30:23.303648+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19461","created_at":"2026-07-05T09:30:23.303648+00:00"},{"alias_kind":"pith_short_12","alias_value":"BEYIZR4EWTBG","created_at":"2026-07-05T09:30:23.303648+00:00"},{"alias_kind":"pith_short_16","alias_value":"BEYIZR4EWTBG3OF5","created_at":"2026-07-05T09:30:23.303648+00:00"},{"alias_kind":"pith_short_8","alias_value":"BEYIZR4E","created_at":"2026-07-05T09:30:23.303648+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31365","citing_title":"Learning to Adapt: Self-Improving Web Agent via Cognitive-Aware Exploration","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z","json":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z.json","graph_json":"https://pith.science/api/pith-number/BEYIZR4EWTBG3OF5DV5UPYE46Z/graph.json","events_json":"https://pith.science/api/pith-number/BEYIZR4EWTBG3OF5DV5UPYE46Z/events.json","paper":"https://pith.science/paper/BEYIZR4E"},"agent_actions":{"view_html":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z","download_json":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z.json","view_paper":"https://pith.science/paper/BEYIZR4E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.19461&json=true","fetch_graph":"https://pith.science/api/pith-number/BEYIZR4EWTBG3OF5DV5UPYE46Z/graph.json","fetch_events":"https://pith.science/api/pith-number/BEYIZR4EWTBG3OF5DV5UPYE46Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z/action/storage_attestation","attest_author":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z/action/author_attestation","sign_citation":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z/action/citation_signature","submit_replication":"https://pith.science/pith/BEYIZR4EWTBG3OF5DV5UPYE46Z/action/replication_record"}},"created_at":"2026-07-05T09:30:23.303648+00:00","updated_at":"2026-07-05T09:30:23.303648+00:00"}