{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HMUKGLANCIKD3UTLCVUWZYRGIO","short_pith_number":"pith:HMUKGLAN","schema_version":"1.0","canonical_sha256":"3b28a32c0d12143dd26b15696ce2264396ecd19b6829fb44254185ae3bdacc21","source":{"kind":"arxiv","id":"2104.08560","version":1},"attestation_state":"computed","paper":{"title":"Mobile App Tasks with Iterative Feedback (MoTIF): Addressing Task Feasibility in Interactive Visual Environments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Andrea Burns, Bryan A. Plummer, Deniz Arsan, Kate Saenko, Ranjitha Kumar, Sanjna Agrawal","submitted_at":"2021-04-17T14:48:02Z","abstract_excerpt":"In recent years, vision-language research has shifted to study tasks which require more complex reasoning, such as interactive question answering, visual common sense reasoning, and question-answer plausibility prediction. However, the datasets used for these problems fail to capture the complexity of real inputs and multimodal environments, such as ambiguous natural language requests and diverse digital domains. We introduce Mobile app Tasks with Iterative Feedback (MoTIF), a dataset with natural language commands for the greatest number of interactive environments to date. MoTIF is the first"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.08560","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-04-17T14:48:02Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"669c66eda14f2f8d648a3152612a3c3e9731fd24f3f8d7ec3edd9c89c72dd3a2","abstract_canon_sha256":"ea396eae1fc2fc1ea6826930b559dacfd013d9fc384f9d826114f77362f24126"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:32:53.392999Z","signature_b64":"QezB5Y3sdMqrSvU+4ses2RqSxwOCRWF1skRA5/AjarguoB+HoGdC3aW2wpBxPyyq8pWW7vUFUIWYK+mLL8rNDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b28a32c0d12143dd26b15696ce2264396ecd19b6829fb44254185ae3bdacc21","last_reissued_at":"2026-07-05T02:32:53.392499Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:32:53.392499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mobile App Tasks with Iterative Feedback (MoTIF): Addressing Task Feasibility in Interactive Visual Environments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Andrea Burns, Bryan A. Plummer, Deniz Arsan, Kate Saenko, Ranjitha Kumar, Sanjna Agrawal","submitted_at":"2021-04-17T14:48:02Z","abstract_excerpt":"In recent years, vision-language research has shifted to study tasks which require more complex reasoning, such as interactive question answering, visual common sense reasoning, and question-answer plausibility prediction. However, the datasets used for these problems fail to capture the complexity of real inputs and multimodal environments, such as ambiguous natural language requests and diverse digital domains. We introduce Mobile app Tasks with Iterative Feedback (MoTIF), a dataset with natural language commands for the greatest number of interactive environments to date. MoTIF is the first"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.08560","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.08560/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.08560","created_at":"2026-07-05T02:32:53.392557+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.08560v1","created_at":"2026-07-05T02:32:53.392557+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.08560","created_at":"2026-07-05T02:32:53.392557+00:00"},{"alias_kind":"pith_short_12","alias_value":"HMUKGLANCIKD","created_at":"2026-07-05T02:32:53.392557+00:00"},{"alias_kind":"pith_short_16","alias_value":"HMUKGLANCIKD3UTL","created_at":"2026-07-05T02:32:53.392557+00:00"},{"alias_kind":"pith_short_8","alias_value":"HMUKGLAN","created_at":"2026-07-05T02:32:53.392557+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12634","citing_title":"MobiBench: Multi-Branch, Modular Benchmark for Mobile GUI Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14573","citing_title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO","json":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO.json","graph_json":"https://pith.science/api/pith-number/HMUKGLANCIKD3UTLCVUWZYRGIO/graph.json","events_json":"https://pith.science/api/pith-number/HMUKGLANCIKD3UTLCVUWZYRGIO/events.json","paper":"https://pith.science/paper/HMUKGLAN"},"agent_actions":{"view_html":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO","download_json":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO.json","view_paper":"https://pith.science/paper/HMUKGLAN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.08560&json=true","fetch_graph":"https://pith.science/api/pith-number/HMUKGLANCIKD3UTLCVUWZYRGIO/graph.json","fetch_events":"https://pith.science/api/pith-number/HMUKGLANCIKD3UTLCVUWZYRGIO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO/action/storage_attestation","attest_author":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO/action/author_attestation","sign_citation":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO/action/citation_signature","submit_replication":"https://pith.science/pith/HMUKGLANCIKD3UTLCVUWZYRGIO/action/replication_record"}},"created_at":"2026-07-05T02:32:53.392557+00:00","updated_at":"2026-07-05T02:32:53.392557+00:00"}