{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QQHROJGIH5MZ6KHSLRVZWLWHVA","short_pith_number":"pith:QQHROJGI","schema_version":"1.0","canonical_sha256":"840f1724c83f599f28f25c6b9b2ec7a8123675ca24a32fbfb9eef22068af1664","source":{"kind":"arxiv","id":"2411.02337","version":3},"attestation_state":"computed","paper":{"title":"WebRL: Training LLM Web Agents via Self-Evolving Online Curriculum Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hanyu Lai, Iat Long Iong, Jiadai Sun, Jie Tang, Shuntian Yao, Tianjie Zhang, Wei Xu, Wenyi Zhao, Xiao Liu, Xinyue Yang, Xueqiao Sun, Yuxiao Dong, Yu Yang, Zehan Qi","submitted_at":"2024-11-04T17:59:58Z","abstract_excerpt":"Large language models (LLMs) have shown remarkable potential as autonomous agents, particularly in web-based tasks. However, existing LLM web agents heavily rely on expensive proprietary LLM APIs, while open LLMs lack the necessary decision-making capabilities. This paper introduces WebRL, a self-evolving online curriculum reinforcement learning framework designed to train high-performance web agents using open LLMs. WebRL addresses three key challenges in building LLM web agents, including the scarcity of training tasks, sparse feedback signals, and policy distribution drift in online learnin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.02337","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-04T17:59:58Z","cross_cats_sorted":[],"title_canon_sha256":"539d87cfb0bcd8350aae34f5c33d8880a6d8efbbeaca92cd76c67d9019f133a8","abstract_canon_sha256":"672dd39bb3610e12acdd6294efbeb33f1eca6ed763d78e47a5eb0e222789ded3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:46.457917Z","signature_b64":"tHT8UUPTgMvaWYlUXSD8Ul4a1osXJUZ9c9wQTE2JHBEnnLKKC2lK/RC+lheJOP5uFiz2YwEPUkrWguJHpzgpBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"840f1724c83f599f28f25c6b9b2ec7a8123675ca24a32fbfb9eef22068af1664","last_reissued_at":"2026-07-05T10:05:46.457412Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:46.457412Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WebRL: Training LLM Web Agents via Self-Evolving Online Curriculum Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hanyu Lai, Iat Long Iong, Jiadai Sun, Jie Tang, Shuntian Yao, Tianjie Zhang, Wei Xu, Wenyi Zhao, Xiao Liu, Xinyue Yang, Xueqiao Sun, Yuxiao Dong, Yu Yang, Zehan Qi","submitted_at":"2024-11-04T17:59:58Z","abstract_excerpt":"Large language models (LLMs) have shown remarkable potential as autonomous agents, particularly in web-based tasks. However, existing LLM web agents heavily rely on expensive proprietary LLM APIs, while open LLMs lack the necessary decision-making capabilities. This paper introduces WebRL, a self-evolving online curriculum reinforcement learning framework designed to train high-performance web agents using open LLMs. WebRL addresses three key challenges in building LLM web agents, including the scarcity of training tasks, sparse feedback signals, and policy distribution drift in online learnin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.02337","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.02337/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.02337","created_at":"2026-07-05T10:05:46.457474+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.02337v3","created_at":"2026-07-05T10:05:46.457474+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.02337","created_at":"2026-07-05T10:05:46.457474+00:00"},{"alias_kind":"pith_short_12","alias_value":"QQHROJGIH5MZ","created_at":"2026-07-05T10:05:46.457474+00:00"},{"alias_kind":"pith_short_16","alias_value":"QQHROJGIH5MZ6KHS","created_at":"2026-07-05T10:05:46.457474+00:00"},{"alias_kind":"pith_short_8","alias_value":"QQHROJGI","created_at":"2026-07-05T10:05:46.457474+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24428","citing_title":"Escaping the Self-Confirmation Trap: An Execute-Distill-Verify Paradigm for Agentic Experience Learning","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21740","citing_title":"Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12485","citing_title":"Speculative Rollback Correction for Quality-Diverse Web Agent Imitation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09447","citing_title":"AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23939","citing_title":"DRIVE: Modeling Skills at the Reasoning and Interaction Levels for Web Agents under Continual Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20291","citing_title":"Weasel: Out-of-Domain Generalization for Web Agents via Importance-Diversity Data Selection","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27209","citing_title":"Learning to Act under Noise: Enhancing Agent Robustness via Noisy Environments","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01091","citing_title":"Deep Research as Rubric for Reinforcement Learning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2504.01990","citing_title":"Advances and Challenges in Foundation Agents: From Brain-Inspired Intelligence to Evolutionary, Collaborative, and Safe Systems","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13727","citing_title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20291","citing_title":"Weasel: Out-of-Domain Generalization for Web Agents via Importance-Diversity Data Selection","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21463","citing_title":"Mem-$\\pi$: Adaptive Memory through Learning When and What to Generate","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20061","citing_title":"Rewarding Beliefs, Not Actions: Consistency-Guided Credit Assignment for Long-Horizon Agents","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03610","citing_title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09572","citing_title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2511.15407","citing_title":"IPR-1: Interactive Physical Reasoner","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22149","citing_title":"DynaWeb: Model-Based Reinforcement Learning of Web Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2603.23964","citing_title":"From Pixels to Digital Agents: An Empirical Study on the Taxonomy and Technological Trends of Reinforcement Learning Environments","ref_index":208,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15841","citing_title":"MEM1: Learning to Synergize Memory and Reasoning for Efficient Long-Horizon Agents","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09423","citing_title":"SimWorld Studio: Automatic Environment Generation with Evolving Coding Agent for Embodied Agent Learning","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09423","citing_title":"SimWorld Studio: Automatic Environment Generation with Evolving Coding Agent for Embodied Agent Learning","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06078","citing_title":"Milestone-Guided Policy Learning for Long-Horizon Language Agents","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00433","citing_title":"Improving LLM Code Generation via Requirement-Aware Curriculum Reinforcement Learning","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06995","citing_title":"What's Missing in Screen-to-Action? Towards a UI-in-the-Loop Paradigm for Multimodal GUI Reasoning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA","json":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA.json","graph_json":"https://pith.science/api/pith-number/QQHROJGIH5MZ6KHSLRVZWLWHVA/graph.json","events_json":"https://pith.science/api/pith-number/QQHROJGIH5MZ6KHSLRVZWLWHVA/events.json","paper":"https://pith.science/paper/QQHROJGI"},"agent_actions":{"view_html":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA","download_json":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA.json","view_paper":"https://pith.science/paper/QQHROJGI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.02337&json=true","fetch_graph":"https://pith.science/api/pith-number/QQHROJGIH5MZ6KHSLRVZWLWHVA/graph.json","fetch_events":"https://pith.science/api/pith-number/QQHROJGIH5MZ6KHSLRVZWLWHVA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA/action/storage_attestation","attest_author":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA/action/author_attestation","sign_citation":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA/action/citation_signature","submit_replication":"https://pith.science/pith/QQHROJGIH5MZ6KHSLRVZWLWHVA/action/replication_record"}},"created_at":"2026-07-05T10:05:46.457474+00:00","updated_at":"2026-07-05T10:05:46.457474+00:00"}