{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:BJJGV4P6G7EN4T5C6LROXGOWLZ","short_pith_number":"pith:BJJGV4P6","schema_version":"1.0","canonical_sha256":"0a526af1fe37c8de4fa2f2e2eb99d65e5528eea9da5293ba4c13a6dd8f3b2316","source":{"kind":"arxiv","id":"2005.03776","version":2},"attestation_state":"computed","paper":{"title":"Mapping Natural Language Instructions to Mobile UI Action Sequences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jason Baldridge, Jiacong He, Xin Zhou, Yang Li, Yuan Zhang","submitted_at":"2020-05-07T21:41:40Z","abstract_excerpt":"We present a new problem: grounding natural language instructions to mobile user interface actions, and create three new datasets for it. For full task evaluation, we create PIXELHELP, a corpus that pairs English instructions with actions performed by people on a mobile UI emulator. To scale training, we decouple the language and action data by (a) annotating action phrase spans in HowTo instructions and (b) synthesizing grounded descriptions of actions for mobile user interfaces. We use a Transformer to extract action phrase tuples from long-range natural language instructions. A grounding Tr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2005.03776","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-05-07T21:41:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2983cdd57c6a7380ec2fad66ee1fd904dc7872aa10f9437ef79c9d57f3c06d6d","abstract_canon_sha256":"5c9268e523d50a166f2f891c5ee8ff1cf38f97ebc9c45f51c60333ba119d9500"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:08:13.132241Z","signature_b64":"gdltgUYw3Xipaj4c9KvgmrbPoD7hNaqtwDZFYPLMTM1QeTBUgngWqx9QAPZt3mx4eybD0ni/Jqy+/B8Oh/YNAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a526af1fe37c8de4fa2f2e2eb99d65e5528eea9da5293ba4c13a6dd8f3b2316","last_reissued_at":"2026-07-05T01:08:13.131791Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:08:13.131791Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mapping Natural Language Instructions to Mobile UI Action Sequences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jason Baldridge, Jiacong He, Xin Zhou, Yang Li, Yuan Zhang","submitted_at":"2020-05-07T21:41:40Z","abstract_excerpt":"We present a new problem: grounding natural language instructions to mobile user interface actions, and create three new datasets for it. For full task evaluation, we create PIXELHELP, a corpus that pairs English instructions with actions performed by people on a mobile UI emulator. To scale training, we decouple the language and action data by (a) annotating action phrase spans in HowTo instructions and (b) synthesizing grounded descriptions of actions for mobile user interfaces. We use a Transformer to extract action phrase tuples from long-range natural language instructions. A grounding Tr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2005.03776","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.03776/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2005.03776","created_at":"2026-07-05T01:08:13.131843+00:00"},{"alias_kind":"arxiv_version","alias_value":"2005.03776v2","created_at":"2026-07-05T01:08:13.131843+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.03776","created_at":"2026-07-05T01:08:13.131843+00:00"},{"alias_kind":"pith_short_12","alias_value":"BJJGV4P6G7EN","created_at":"2026-07-05T01:08:13.131843+00:00"},{"alias_kind":"pith_short_16","alias_value":"BJJGV4P6G7EN4T5C","created_at":"2026-07-05T01:08:13.131843+00:00"},{"alias_kind":"pith_short_8","alias_value":"BJJGV4P6","created_at":"2026-07-05T01:08:13.131843+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04227","citing_title":"Mobile GUI Agents under Real-world Threats: Are We There Yet?","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2401.10935","citing_title":"SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10458","citing_title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07972","citing_title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ","json":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ.json","graph_json":"https://pith.science/api/pith-number/BJJGV4P6G7EN4T5C6LROXGOWLZ/graph.json","events_json":"https://pith.science/api/pith-number/BJJGV4P6G7EN4T5C6LROXGOWLZ/events.json","paper":"https://pith.science/paper/BJJGV4P6"},"agent_actions":{"view_html":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ","download_json":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ.json","view_paper":"https://pith.science/paper/BJJGV4P6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2005.03776&json=true","fetch_graph":"https://pith.science/api/pith-number/BJJGV4P6G7EN4T5C6LROXGOWLZ/graph.json","fetch_events":"https://pith.science/api/pith-number/BJJGV4P6G7EN4T5C6LROXGOWLZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ/action/storage_attestation","attest_author":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ/action/author_attestation","sign_citation":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ/action/citation_signature","submit_replication":"https://pith.science/pith/BJJGV4P6G7EN4T5C6LROXGOWLZ/action/replication_record"}},"created_at":"2026-07-05T01:08:13.131843+00:00","updated_at":"2026-07-05T01:08:13.131843+00:00"}