{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JWKLYADBTC3BCOBGWLSGY35LIC","short_pith_number":"pith:JWKLYADB","schema_version":"1.0","canonical_sha256":"4d94bc006198b6113826b2e46c6fab4086072c7b08e4717909fe3e7192aef415","source":{"kind":"arxiv","id":"2410.14803","version":5},"attestation_state":"computed","paper":{"title":"DistRL: An Asynchronous Distributed Reinforcement Learning Framework for On-Device Control Agents","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Jianheng Liu, Jianye Hao, Jun Wang, Kun Shao, Taiyi Wang, Zhihao Wu","submitted_at":"2024-10-18T18:19:56Z","abstract_excerpt":"On-device control agents, especially on mobile devices, are responsible for operating mobile devices to fulfill users' requests, enabling seamless and intuitive interactions. Integrating Multimodal Large Language Models (MLLMs) into these agents enhances their ability to understand and execute complex commands, thereby improving user experience. However, fine-tuning MLLMs for on-device control presents significant challenges due to limited data availability and inefficient online training processes. This paper introduces DistRL, a novel framework designed to enhance the efficiency of online RL"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14803","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-18T18:19:56Z","cross_cats_sorted":["cs.AI","cs.DC","cs.SY","eess.SY"],"title_canon_sha256":"7cd82a33e7d6b9ce507f720bed883783660ce03e1a300a757216d29f434fb834","abstract_canon_sha256":"43b6b8a5a244fcffd8259359aa069d8bc5a92e8f81b4412fbc491010d5aa391d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:50.850062Z","signature_b64":"IgTLl8FyZHHzlz99dj+0dWllz9wU9zwwESuAUTWGFwXpddulDHPmMTiYzbrrG8Hn2d2vN0MaqyVF2RmjcjoSAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d94bc006198b6113826b2e46c6fab4086072c7b08e4717909fe3e7192aef415","last_reissued_at":"2026-07-05T10:17:50.849576Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:50.849576Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DistRL: An Asynchronous Distributed Reinforcement Learning Framework for On-Device Control Agents","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Jianheng Liu, Jianye Hao, Jun Wang, Kun Shao, Taiyi Wang, Zhihao Wu","submitted_at":"2024-10-18T18:19:56Z","abstract_excerpt":"On-device control agents, especially on mobile devices, are responsible for operating mobile devices to fulfill users' requests, enabling seamless and intuitive interactions. Integrating Multimodal Large Language Models (MLLMs) into these agents enhances their ability to understand and execute complex commands, thereby improving user experience. However, fine-tuning MLLMs for on-device control presents significant challenges due to limited data availability and inefficient online training processes. This paper introduces DistRL, a novel framework designed to enhance the efficiency of online RL"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14803","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14803/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14803","created_at":"2026-07-05T10:17:50.849640+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14803v5","created_at":"2026-07-05T10:17:50.849640+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14803","created_at":"2026-07-05T10:17:50.849640+00:00"},{"alias_kind":"pith_short_12","alias_value":"JWKLYADBTC3B","created_at":"2026-07-05T10:17:50.849640+00:00"},{"alias_kind":"pith_short_16","alias_value":"JWKLYADBTC3BCOBG","created_at":"2026-07-05T10:17:50.849640+00:00"},{"alias_kind":"pith_short_8","alias_value":"JWKLYADB","created_at":"2026-07-05T10:17:50.849640+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07027","citing_title":"StainFlow: Entity-Stain Tracking and Evidence Linking for Process Rewards in GUI Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":273,"is_internal_anchor":false},{"citing_arxiv_id":"2506.17697","citing_title":"Beyond Syntax: Action Semantics Learning for App Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":249,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC","json":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC.json","graph_json":"https://pith.science/api/pith-number/JWKLYADBTC3BCOBGWLSGY35LIC/graph.json","events_json":"https://pith.science/api/pith-number/JWKLYADBTC3BCOBGWLSGY35LIC/events.json","paper":"https://pith.science/paper/JWKLYADB"},"agent_actions":{"view_html":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC","download_json":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC.json","view_paper":"https://pith.science/paper/JWKLYADB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14803&json=true","fetch_graph":"https://pith.science/api/pith-number/JWKLYADBTC3BCOBGWLSGY35LIC/graph.json","fetch_events":"https://pith.science/api/pith-number/JWKLYADBTC3BCOBGWLSGY35LIC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC/action/storage_attestation","attest_author":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC/action/author_attestation","sign_citation":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC/action/citation_signature","submit_replication":"https://pith.science/pith/JWKLYADBTC3BCOBGWLSGY35LIC/action/replication_record"}},"created_at":"2026-07-05T10:17:50.849640+00:00","updated_at":"2026-07-05T10:17:50.849640+00:00"}