{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3XM7CB6FR6XV52BSDRBPMI57WE","short_pith_number":"pith:3XM7CB6F","schema_version":"1.0","canonical_sha256":"ddd9f107c58faf5ee8321c42f623bfb10e5957db36b6302e2e2bf044b3c49c24","source":{"kind":"arxiv","id":"2406.11896","version":1},"attestation_state":"computed","paper":{"title":"DigiRL: Training In-The-Wild Device-Control Agents with Autonomous Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alane Suhr, Aviral Kumar, Hao Bai, Jiayi Pan, Mert Cemri, Sergey Levine, Yifei Zhou","submitted_at":"2024-06-14T17:49:55Z","abstract_excerpt":"Training corpuses for vision language models (VLMs) typically lack sufficient amounts of decision-centric data. This renders off-the-shelf VLMs sub-optimal for decision-making tasks such as in-the-wild device control through graphical user interfaces (GUIs). While training with static demonstrations has shown some promise, we show that such methods fall short for controlling real GUIs due to their failure to deal with real-world stochasticity and non-stationarity not captured in static observational data. This paper introduces a novel autonomous RL approach, called DigiRL, for training in-the-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11896","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-14T17:49:55Z","cross_cats_sorted":[],"title_canon_sha256":"ce1dae3f371c6776216cc0df534245b6e4fe1869ee8a4b991d26b0c96e170d3b","abstract_canon_sha256":"bd649dd4f7baaa804d3b2c658e2f430307b7f451a100d92033b290a6f0bc9712"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:52.187535Z","signature_b64":"4vkYcztuLtNCDb7x1DA6/ZBwk3Hu29cBv5w8Zv8MAa+BLGiL9cI3gnD4SNsL5HKCMeGaKDIcHB5IR3fXmXseAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddd9f107c58faf5ee8321c42f623bfb10e5957db36b6302e2e2bf044b3c49c24","last_reissued_at":"2026-07-05T08:33:52.187067Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:52.187067Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DigiRL: Training In-The-Wild Device-Control Agents with Autonomous Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alane Suhr, Aviral Kumar, Hao Bai, Jiayi Pan, Mert Cemri, Sergey Levine, Yifei Zhou","submitted_at":"2024-06-14T17:49:55Z","abstract_excerpt":"Training corpuses for vision language models (VLMs) typically lack sufficient amounts of decision-centric data. This renders off-the-shelf VLMs sub-optimal for decision-making tasks such as in-the-wild device control through graphical user interfaces (GUIs). While training with static demonstrations has shown some promise, we show that such methods fall short for controlling real GUIs due to their failure to deal with real-world stochasticity and non-stationarity not captured in static observational data. This paper introduces a novel autonomous RL approach, called DigiRL, for training in-the-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11896","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11896/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11896","created_at":"2026-07-05T08:33:52.187132+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11896v1","created_at":"2026-07-05T08:33:52.187132+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11896","created_at":"2026-07-05T08:33:52.187132+00:00"},{"alias_kind":"pith_short_12","alias_value":"3XM7CB6FR6XV","created_at":"2026-07-05T08:33:52.187132+00:00"},{"alias_kind":"pith_short_16","alias_value":"3XM7CB6FR6XV52BS","created_at":"2026-07-05T08:33:52.187132+00:00"},{"alias_kind":"pith_short_8","alias_value":"3XM7CB6F","created_at":"2026-07-05T08:33:52.187132+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09447","citing_title":"AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29537","citing_title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06322","citing_title":"DragOn: A Benchmark and Dataset for Drag-Based GUI Interactions","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2504.01990","citing_title":"Advances and Challenges in Foundation Agents: From Brain-Inspired Intelligence to Evolutionary, Collaborative, and Safe Systems","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05765","citing_title":"X-OmniClaw Technical Report: A Unified Mobile Agent for Multimodal Understanding and Interaction","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":272,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09572","citing_title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05765","citing_title":"X-OmniClaw Technical Report: A Unified Mobile Agent for Multimodal Understanding and Interaction","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15607","citing_title":"Imperfectly Cooperative Human-AI Interactions: Comparing the Impacts of Human and AI Attributes in Simulated and User Studies","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE","json":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE.json","graph_json":"https://pith.science/api/pith-number/3XM7CB6FR6XV52BSDRBPMI57WE/graph.json","events_json":"https://pith.science/api/pith-number/3XM7CB6FR6XV52BSDRBPMI57WE/events.json","paper":"https://pith.science/paper/3XM7CB6F"},"agent_actions":{"view_html":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE","download_json":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE.json","view_paper":"https://pith.science/paper/3XM7CB6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11896&json=true","fetch_graph":"https://pith.science/api/pith-number/3XM7CB6FR6XV52BSDRBPMI57WE/graph.json","fetch_events":"https://pith.science/api/pith-number/3XM7CB6FR6XV52BSDRBPMI57WE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE/action/storage_attestation","attest_author":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE/action/author_attestation","sign_citation":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE/action/citation_signature","submit_replication":"https://pith.science/pith/3XM7CB6FR6XV52BSDRBPMI57WE/action/replication_record"}},"created_at":"2026-07-05T08:33:52.187132+00:00","updated_at":"2026-07-05T08:33:52.187132+00:00"}