{"as_of":"2026-08-06T02:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4b51c827b09a5eb98efa9667df1631b67040d2c317b0701056c0b587ee62c2b8","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":108,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T21:01:24.792889Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2504.12501","last_updated":"2026-08-03T01:47:58Z","snapshot_observed_at":"2026-08-06T02:15:42.974365Z","submitted_at":"2025-04-16T21:36:46Z","title":"Reinforcement Learning from Human Feedback","version":9},"reference_index":189,"source":"pdf_text","source_observed_at":"2026-05-22T19:27:40.991325Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2504.12501"},"observation_digest":"sha256:caf80376459b38324a62099c7cd569ebc698ad0ef69b19cc284935edfeb3b2f5","observation_id":"09feafe4-9c86-4aab-bd19-aa4064a5e969","resolution":{"observed_at":"2026-05-22T19:32:01.417888Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2504.21776","last_updated":"2025-10-13T12:40:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-30T16:25:25Z","title":"WebThinker: Empowering Large Reasoning Models with Deep Research Capability","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-16T19:14:25.283645Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2504.21776"},"observation_digest":"sha256:7b8dd03df10ea59be9ca9cae53dd71ab02df5014d6a60d54ed37c1071ba01792","observation_id":"6f662b05-6ca3-490d-8994-c6cef5f2f129","resolution":{"observed_at":"2026-05-16T19:14:25.317621Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2505.10978","last_updated":"2025-10-28T15:11:36Z","snapshot_observed_at":"2026-07-29T19:20:21.974239Z","submitted_at":"2025-05-16T08:26:59Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-11T09:15:08.193357Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2505.10978"},"observation_digest":"sha256:b2c83c7389d5ce87d7943b2b7329cd9691b820665386f25c577ab5efc3a43069","observation_id":"030595a1-8013-4c92-8910-92ee6bb51497","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2505.18719","last_updated":"2025-05-24T14:42:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-24T14:42:51Z","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:40.245908Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2505.18719"},"observation_digest":"sha256:20bdfd0f7b515afb12808e0a40b7ebeac128573e19459656a5a81b01239fd965","observation_id":"5ab7702e-0e98-4869-a4c7-6d183698e938","resolution":{"observed_at":"2026-05-16T12:55:40.342411Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2505.23678","last_updated":"2026-05-15T17:27:46Z","snapshot_observed_at":"2026-08-02T04:30:21.704758Z","submitted_at":"2025-05-29T17:20:26Z","title":"Grounded Reinforcement Learning for Visual Reasoning","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-22T01:05:18.801388Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2505.23678"},"observation_digest":"sha256:a0f17c173af630b74fd42aee39051bc50ee806a640437d2a3bfc6ce725c2c1c9","observation_id":"85946968-b8e4-49d9-a7fc-ae0c82e95606","resolution":{"observed_at":"2026-05-22T01:05:51.865184Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:582d4bab7698bf343c19859b310300e1fd5c1e8cece56b4754a7ad3bc4e944b8","observation_id":"a31baf8e-8ee3-419f-a2ea-3f0aba9b8c7a","resolution":{"observed_at":"2026-05-14T22:23:15.493799Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2508.07407","last_updated":"2025-08-31T14:55:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-10T16:07:32Z","title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","version":2},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-05-15T23:21:42.029285Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2508.07407"},"observation_digest":"sha256:99250f5adcc73b27af6fe7bbeebc7ade05f8879e7eaa7edf5fd2005ab76eff3d","observation_id":"691ecce4-10a8-4e84-81de-4c7146df54da","resolution":{"observed_at":"2026-05-15T23:21:42.174087Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T21:01:24.792889Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09586","last_updated":"2025-08-20T07:50:49Z","snapshot_observed_at":"2026-08-05T21:01:22.543051Z","submitted_at":"2025-08-13T07:59:29Z","title":"EvoCurr: Self-evolving Curriculum with Behavior Code Generation for Complex Decision-making","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T21:01:24.792889Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2508.09586"},"observation_digest":"sha256:2c7fff8aafb0a2e12f435e41d78c79db2f5222dbf1cbf5c2a8f26b1ffcbbf7bc","observation_id":"c9593914-92a1-4285-8e4e-cfd5c89eafb6","resolution":{"observed_at":"2026-08-05T21:01:24.792889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T20:20:56.119890Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.10839","last_updated":"2025-08-14T17:05:44Z","snapshot_observed_at":"2026-08-05T20:20:45.156819Z","submitted_at":"2025-08-14T17:05:44Z","title":"Reinforced Language Models for Sequential Decision Making","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T20:20:56.119890Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2508.10839"},"observation_digest":"sha256:46ea5f2835dc9a3261ea5ef121c434aa984fdac80baa3cb8c55c721e164832bd","observation_id":"8aaccf3a-20ba-495a-88f2-cf358bd7790b","resolution":{"observed_at":"2026-08-05T20:20:56.119890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T20:17:12.728962Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning, 2025 b","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.10874","last_updated":"2025-08-14T17:46:01Z","snapshot_observed_at":"2026-08-05T20:16:58.323930Z","submitted_at":"2025-08-14T17:46:01Z","title":"SSRL: Self-Search Reinforcement Learning","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-05T20:17:12.728962Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2508.10874"},"observation_digest":"sha256:c3c8c59676a476c5227e15343eec398244bdb1638ca1ca3875d0490ebf95c338","observation_id":"ac04a7b9-7d04-41c4-be28-29de60856bbc","resolution":{"observed_at":"2026-08-05T20:17:12.728962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T18:55:33.722891Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.13787","last_updated":"2025-08-19T12:43:49Z","snapshot_observed_at":"2026-08-05T18:55:30.913690Z","submitted_at":"2025-08-19T12:43:49Z","title":"BetaWeb: Towards a Blockchain-enabled Trustworthy Agentic Web","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-05T18:55:33.722891Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2508.13787"},"observation_digest":"sha256:bd1bd9d98a2944488420deb3969a531b7ac5c98a7e247e2584cd63f6ab2ba206","observation_id":"324d735f-e02e-433e-9b4a-3469afc45f9a","resolution":{"observed_at":"2026-08-05T18:55:33.722891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T16:22:51.511495Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.18669","last_updated":"2025-08-26T04:26:29Z","snapshot_observed_at":"2026-08-05T16:22:50.856434Z","submitted_at":"2025-08-26T04:26:29Z","title":"MUA-RL: Multi-turn User-interacting Agent Reinforcement Learning for agentic tool use","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T16:22:51.511495Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2508.18669"},"observation_digest":"sha256:19f025742a422dfaa46d56598b8f64a9a27fe249e5632562d853c7c85323f601","observation_id":"20ea23a1-92a8-4a08-8f9b-c2c8a16bfe37","resolution":{"observed_at":"2026-08-05T16:22:51.511495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T15:44:14.910639Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.19598","last_updated":"2025-08-27T06:19:50Z","snapshot_observed_at":"2026-08-05T15:44:13.449330Z","submitted_at":"2025-08-27T06:19:50Z","title":"Encouraging Good Processes Without the Need for Good Answers: Reinforcement Learning for LLM Agent Planning","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T15:44:14.910639Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2508.19598"},"observation_digest":"sha256:b561432f955fc442cc0ffe820934210366095e3e3d2bb3081a99485cc467e7e3","observation_id":"36916772-8f61-48ec-b5b7-bf8d65ed929e","resolution":{"observed_at":"2026-08-05T15:44:14.910639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T11:39:24.770094Z","title":"RAGEN: Understanding Self-evolution in LLM Agents via Multi-turn Reinforcement Learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02479","last_updated":"2025-09-03T17:06:42Z","snapshot_observed_at":"2026-08-05T11:39:17.950056Z","submitted_at":"2025-09-02T16:30:19Z","title":"SimpleTIR: End-to-End Reinforcement Learning for Multi-Turn Tool-Integrated Reasoning","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T11:39:24.770094Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2509.02479"},"observation_digest":"sha256:fa026056b73b339f8ac82417577462d4a0ddd5a1410b1542d4a058dd8f607c87","observation_id":"2beff123-71bc-4bc3-bd4c-76dbac3f7860","resolution":{"observed_at":"2026-08-05T11:39:24.770094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2509.02547","last_updated":"2026-04-17T18:09:08Z","snapshot_observed_at":"2026-08-03T09:07:42.489237Z","submitted_at":"2025-09-02T17:46:26Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","version":5},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-18T19:19:36.427337Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2509.02547"},"observation_digest":"sha256:9496ddc5f3ef29b34ad7e9be59b677c9c06e3bf13da4e4fa2d21b91ec40932bf","observation_id":"4e4c5d38-576b-494f-b239-26b6df0b04bd","resolution":{"observed_at":"2026-05-18T19:21:48.689674Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-04T16:07:43.345147Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-04T16:07:24.699834Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"reference_index":191,"source":"pdf_text","source_observed_at":"2026-08-04T16:07:43.345147Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2509.16679"},"observation_digest":"sha256:5de759e4ad9ef31c1cef252a47417341bad57682fcb758d5ed5cd2dbc5eee0ba","observation_id":"c0e8479f-58f0-426e-b168-7fa18833324b","resolution":{"observed_at":"2026-08-04T16:07:43.345147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-04T10:48:04.269782Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.08558","last_updated":"2026-05-24T21:56:11Z","snapshot_observed_at":"2026-08-04T10:47:54.049813Z","submitted_at":"2025-10-09T17:59:17Z","title":"Agent Learning via Early Experience","version":3},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-04T10:48:04.269782Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2510.08558"},"observation_digest":"sha256:4fe5323d768a1eab01010571ddcb62e2455127d0666040b22a39ba7252e5280c","observation_id":"2f2a474c-08e4-4498-84cb-d20a478138f9","resolution":{"observed_at":"2026-08-04T10:48:04.269782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2510.13727","last_updated":"2026-05-19T04:39:26Z","snapshot_observed_at":"2026-08-02T19:37:00.796741Z","submitted_at":"2025-10-15T16:30:57Z","title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-21T20:42:40.823721Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2510.13727"},"observation_digest":"sha256:87b60e0f4e9b6afb85c0370707f1ff6ea98811b1c108dc93a972907acec23162","observation_id":"2bbf6ac5-1307-49ac-bbbe-a979c670fcf3","resolution":{"observed_at":"2026-05-21T20:44:22.027160Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2510.22977","last_updated":"2026-04-17T17:15:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-27T03:58:29Z","title":"The Reasoning Trap: How Enhancing LLM Reasoning Amplifies Tool Hallucination","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T04:09:50.183494Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2510.22977"},"observation_digest":"sha256:84aabc404693b570d8ec1dfcd8c1835d6c4531da50be044715c4c923aa315faf","observation_id":"b552fccc-256f-4aa6-8cef-bdfd395e92f1","resolution":{"observed_at":"2026-05-18T04:10:51.312197Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-04T07:17:44.658592Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.26270","last_updated":"2026-05-28T11:04:33Z","snapshot_observed_at":"2026-08-04T16:18:16.535621Z","submitted_at":"2025-10-30T08:53:41Z","title":"Graph-Enhanced Policy Optimization in LLM Agent Training","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T07:17:44.658592Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2510.26270"},"observation_digest":"sha256:c867164a2fd3b90b825fac1c38654c0407953e567e184539aca3453f268727d1","observation_id":"7c05bf3a-330b-487b-8dbb-4d062324d8c2","resolution":{"observed_at":"2026-08-04T07:17:44.658592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-03T18:03:38.550009Z","title":"N., Liu, L., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.07287","last_updated":"2026-06-28T11:58:42Z","snapshot_observed_at":"2026-08-05T13:20:26.960914Z","submitted_at":"2025-12-08T08:27:24Z","title":"Experience-Evolving Multi-Turn Tool-Use Agent with Hybrid Episodic-Procedural Memory","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T18:03:38.550009Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2512.07287"},"observation_digest":"sha256:fb10a89138dce5ad78a2d2d76ff5eebb76169db7335815f7c77b44216d20cb28","observation_id":"59563790-729f-4fc6-a6d7-c7168d26fa22","resolution":{"observed_at":"2026-08-03T18:03:38.550009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-03T16:30:42.182243Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.13278","last_updated":"2026-06-05T03:56:56Z","snapshot_observed_at":"2026-08-03T16:30:33.694000Z","submitted_at":"2025-12-15T12:38:04Z","title":"AutoTool: Dynamic Tool Selection and Integration for Agentic Reasoning","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-03T16:30:42.182243Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2512.13278"},"observation_digest":"sha256:04b13db2d16b81320f0d801ee8ba0bff4ed7d9730e9d05c0a9ae549c58a0c235","observation_id":"9f29736f-0b7a-451f-8e22-4bbfdb37d26e","resolution":{"observed_at":"2026-08-03T16:30:42.182243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2512.17102","last_updated":"2026-03-10T05:40:56Z","snapshot_observed_at":"2026-08-02T15:17:10.009854Z","submitted_at":"2025-12-18T21:58:19Z","title":"Reinforcement Learning for Self-Improving Agent with Skill Library","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T20:05:30.593804Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2512.17102"},"observation_digest":"sha256:4faa5040854f57ef7a5b217c668f59592d460c2575aabf8b1bafc83de02e7eed","observation_id":"b9f191fc-e0ae-4a81-8328-6c2d48299049","resolution":{"observed_at":"2026-05-17T20:05:30.610948Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2602.02556","last_updated":"2026-04-24T18:39:50Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-01-30T13:15:13Z","title":"Beyond Experience Retrieval: Learning to Generate Utility-Optimized Structured Experience for Frozen LLMs","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T09:43:55.255948Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2602.02556"},"observation_digest":"sha256:8b67c3bc2dc16e53efebbd5433473b40bbc5e30d9bbfaf201bf3f2a87ff36f08","observation_id":"ec99d046-96b7-449d-9fbb-9947731020c9","resolution":{"observed_at":"2026-05-16T09:47:41.958456Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2602.16699","last_updated":"2026-05-15T21:12:58Z","snapshot_observed_at":"2026-07-06T22:46:21.571641Z","submitted_at":"2026-02-18T18:46:14Z","title":"Calibrate-Then-Act: Cost-Aware Exploration in LLM Agents","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2602.16699"},"observation_digest":"sha256:4d7f786383ec1ada7867e408d638bc64ab199e097e66e979ebd1b3582aff5e97","observation_id":"0ef3f9e4-a74f-43ac-b707-a85ec80bcedb","resolution":{"observed_at":"2026-05-21T12:40:08.627745Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2603.00977","last_updated":"2026-05-05T03:37:16Z","snapshot_observed_at":"2026-08-02T14:44:32.701926Z","submitted_at":"2026-03-01T08:09:03Z","title":"HiMAC: Hierarchical Macro-Micro Learning for Long-Horizon LLM Agents","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-15T18:35:45.900606Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2603.00977"},"observation_digest":"sha256:112de293a1bba2a361cf832b4e0889a99feebb5c89adc538201998bb568379e2","observation_id":"ece3d37d-2f0d-4447-80b0-1c334b7bf0f2","resolution":{"observed_at":"2026-05-15T18:36:28.338027Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2603.24709","last_updated":"2026-04-06T20:50:24Z","snapshot_observed_at":"2026-07-06T22:50:32.832499Z","submitted_at":"2026-03-25T18:31:39Z","title":"Training LLMs for Multi-Step Tool Orchestration with Constrained Data Synthesis and Graduated Rewards","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T00:15:56.194714Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2603.24709"},"observation_digest":"sha256:982baa399479b862ca586a9c61a44f78ddc3936ec2fdc047ea6a3f0eac521c03","observation_id":"44f760fe-5aac-4d76-8b02-ba402f612dc9","resolution":{"observed_at":"2026-05-15T00:18:22.746285Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.03675","last_updated":"2026-05-23T06:41:17Z","snapshot_observed_at":"2026-08-03T22:30:44.538626Z","submitted_at":"2026-04-04T10:23:46Z","title":"OASES: Outcome-Aligned Search-Evaluation Co-Training for Agentic Search","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-13T17:16:09.927267Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.03675"},"observation_digest":"sha256:1cb6b44a88285614ee1a494c2a1105e7d1bede3192c093aeaa28cecbb142066d","observation_id":"b31f1fa8-1c7f-41f1-8d3c-5765fff0547f","resolution":{"observed_at":"2026-05-13T17:16:44.180135Z","resolver_source":"orphan_title_repair","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.06777","last_updated":"2026-04-08T07:48:07Z","snapshot_observed_at":"2026-08-05T03:57:17.041961Z","submitted_at":"2026-04-08T07:48:07Z","title":"Walk the Talk: Bridging the Reasoning-Action Gap for Thinking with Images via Multimodal Agentic Policy Optimization","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-10T18:20:02.559108Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.06777"},"observation_digest":"sha256:f04fafde1c28f922a201f7b44314e5c2f77abd1123268dd9578159eeeb6d9ab7","observation_id":"943c6789-6844-4526-b83a-c2c61459f484","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.09455","last_updated":"2026-04-10T16:14:48Z","snapshot_observed_at":"2026-07-06T22:58:21.624968Z","submitted_at":"2026-04-10T16:14:48Z","title":"E3-TIR: Enhanced Experience Exploitation for Tool-Integrated Reasoning","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T18:08:18.056525Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.09455"},"observation_digest":"sha256:cbcf7825b43f366990d26d2464f73a4180b8506bdcaae1a6d1dfcceaf6a3f66a","observation_id":"15d605c7-bc36-4672-b51c-0984e9909ea1","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.09459","last_updated":"2026-04-13T12:08:22Z","snapshot_observed_at":"2026-07-06T22:58:21.624968Z","submitted_at":"2026-04-10T16:17:44Z","title":"From Reasoning to Agentic: Credit Assignment in Reinforcement Learning for Large Language Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T17:09:36.341574Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.09459"},"observation_digest":"sha256:22bebb35f5043c5b0a86dac81bd1925d730463facea57048a65bfd476a7540e1","observation_id":"644bb98b-0882-48c1-ab52-9181e316eb54","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.10674","last_updated":"2026-04-12T14:57:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-12T14:57:52Z","title":"Skill-SD: Skill-Conditioned Self-Distillation for Multi-turn LLM Agents","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-10T15:28:07.981488Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.10674"},"observation_digest":"sha256:b5a423a5480988e7818e037dab6c056169c22f3852f8b214b0ceed67cc313cfc","observation_id":"983c08d9-220c-45e6-9d95-2ae8b5048398","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.14518","last_updated":"2026-04-17T01:58:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-16T01:20:06Z","title":"Mind DeepResearch Technical Report","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T11:46:49.178896Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.14518"},"observation_digest":"sha256:ed0155430a582641c45a2f91decb0bb80ef4cc6c73241344c37f47a09ce77cac","observation_id":"655764ac-ffe7-46c5-aabe-961de77ac732","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.18131","last_updated":"2026-04-20T11:54:20Z","snapshot_observed_at":"2026-07-06T23:05:04.279268Z","submitted_at":"2026-04-20T11:54:20Z","title":"Training LLM Agents for Spontaneous, Reward-Free Self-Evolution via World Knowledge Exploration","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T04:36:27.381942Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.18131"},"observation_digest":"sha256:cb06c59e432aeffe4692fe0b7d82aee864d5560a94a80bc0b8ce94bdb9d32f17","observation_id":"c5e44bd2-d50e-4964-b7d9-a0d1b125defe","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.18133","last_updated":"2026-04-20T12:00:31Z","snapshot_observed_at":"2026-08-03T09:21:05.928583Z","submitted_at":"2026-04-20T12:00:31Z","title":"Multi-Agent Systems: From Classical Paradigms to Large Foundation Model-Enabled Futures","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-10T04:31:28.242097Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.18133"},"observation_digest":"sha256:6edfd61bf1e9576d963dd958f3711cbaf3bc8a86402361f166e2c2d6fbe2520e","observation_id":"f4cf9952-ad33-4e9f-8c30-89926b88acc9","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.18401","last_updated":"2026-06-21T13:04:23Z","snapshot_observed_at":"2026-07-06T23:05:17.639285Z","submitted_at":"2026-04-20T15:22:39Z","title":"StepPO: Step-Aligned Policy Optimization for Agentic Reinforcement Learning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T04:26:59.781416Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.18401"},"observation_digest":"sha256:b939b16ee221f958514cf484a3540042fbcb71802989644a9ef03e333f6bc67b","observation_id":"9bceaa4a-1cf9-40e0-898b-a2fd027ee004","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.18401","last_updated":"2026-06-21T13:04:23Z","snapshot_observed_at":"2026-07-06T23:05:17.639285Z","submitted_at":"2026-04-20T15:22:39Z","title":"StepPO: Step-Aligned Policy Optimization for Agentic Reinforcement Learning","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-05T12:23:00.487399Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.18401"},"observation_digest":"sha256:399cf28ee561976e0a62f1a9f57203171a9dc00d6c488af1a65cb2326a838077","observation_id":"dcbea456-1568-4b82-bd7f-8fa34f17f7c9","resolution":{"observed_at":"2026-07-05T12:30:59.971516Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.19485","last_updated":"2026-04-21T14:07:39Z","snapshot_observed_at":"2026-07-06T23:06:07.161558Z","submitted_at":"2026-04-21T14:07:39Z","title":"EVPO: Explained Variance Policy Optimization for Adaptive Critic Utilization in LLM Post-Training","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T03:00:10.404757Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.19485"},"observation_digest":"sha256:939811d0c338bf377eb852575ea9fa631bdb9e968b733f1dd10f4f2c2fdeddd1","observation_id":"f2bb9c72-46d1-47e0-917c-46b24afd5490","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.19656","last_updated":"2026-04-21T16:45:29Z","snapshot_observed_at":"2026-07-06T23:06:16.972438Z","submitted_at":"2026-04-21T16:45:29Z","title":"Pause or Fabricate? Training Language Models for Grounded Reasoning","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-10T03:01:58.366028Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.19656"},"observation_digest":"sha256:c8b21b11bc9968e9f8c5c106b97e8cffeae5749a9d684f642c5bf0dc9828bc60","observation_id":"b28fdab9-0c1a-4183-8774-625d16ffb86d","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.22452","last_updated":"2026-04-24T11:11:36Z","snapshot_observed_at":"2026-07-06T23:08:51.913995Z","submitted_at":"2026-04-24T11:11:36Z","title":"Superminds Test: Actively Evaluating Collective Intelligence of Agent Society via Probing Agents","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-08T12:00:07.345611Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.22452"},"observation_digest":"sha256:77afccd6088ccc371a22fac3a622cd11f0810b4cb80df35ae5f5b74934c43862","observation_id":"ed0ef21a-ed4b-423a-8c7f-c380277a5ff0","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2604.23838","last_updated":"2026-04-26T18:45:31Z","snapshot_observed_at":"2026-08-02T18:30:10.398959Z","submitted_at":"2026-04-26T18:45:31Z","title":"JigsawRL: Assembling RL Pipelines for Efficient LLM Post-Training","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-08T06:36:33.914193Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2604.23838"},"observation_digest":"sha256:de2363cd2eef6aced4d9ec172e11db2444272b78687b9bf736a4a3468005ec94","observation_id":"a1294068-88c4-4bce-9250-de59386412c3","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.00347","last_updated":"2026-05-01T02:05:56Z","snapshot_observed_at":"2026-07-06T23:13:47.082550Z","submitted_at":"2026-05-01T02:05:56Z","title":"Odysseus: Scaling VLMs to 100+ Turn Decision-Making in Games via Reinforcement Learning","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-09T20:22:58.061772Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.00347"},"observation_digest":"sha256:d62ca9ff27eef25fef17b5818abed101c0c02c46586cd7c0ae94e0fe32b275ee","observation_id":"921c004b-82bb-47f0-801d-b557f01cc9b5","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.02178","last_updated":"2026-05-04T03:15:56Z","snapshot_observed_at":"2026-08-02T07:53:28.208356Z","submitted_at":"2026-05-04T03:15:56Z","title":"T$^2$PO: Uncertainty-Guided Exploration Control for Stable Multi-Turn Agentic Reinforcement Learning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-08T19:36:00.359351Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.02178"},"observation_digest":"sha256:2c84113fd56ffbd22d8354da9e91e921dbdf1083ddf31917dc6ed8e954005c25","observation_id":"7b56134a-cb9a-4da6-b448-033bfae798d1","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.02913","last_updated":"2026-04-08T00:53:29Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-08T00:53:29Z","title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","version":1},"reference_index":126,"source":"arxiv_source","source_observed_at":"2026-05-10T19:15:27.406778Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.02913"},"observation_digest":"sha256:dc151c0f4683a3e7a9275df44b093eca8d654e3ccbb7fa4df84f5c594167a734","observation_id":"ab820f4d-5ba5-451b-ba60-7dbe9186081e","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.06130","last_updated":"2026-05-12T10:25:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-07T12:33:30Z","title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-08T10:23:52.522238Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.06130"},"observation_digest":"sha256:5ba4743ae04c8cb841a3927f2e3c1ac33f470ae38edd703ba44813c0a1853b1f","observation_id":"b18275cc-3954-4200-9074-b5d16f1b970f","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.06130","last_updated":"2026-05-12T10:25:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-07T12:33:30Z","title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-11T02:00:00.663355Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.06130"},"observation_digest":"sha256:3e639927f9e4827b35d4befc881fbc006e1ac4cd8c8dcbfd56953a7d817c7f70","observation_id":"4a4a1fc4-50df-43db-a8bb-60dfe1286238","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.06130","last_updated":"2026-05-12T10:25:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-07T12:33:30Z","title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","version":3},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-13T07:17:13.708752Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.06130"},"observation_digest":"sha256:b544d730c78b2c35c05fb2362a026c136e19046c4fdb1891bb185adffdc61a1b","observation_id":"14c0c2fd-2909-42dc-9d5b-49ec524756e9","resolution":{"observed_at":"2026-05-13T07:17:28.448889Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.06200","last_updated":"2026-05-07T13:09:31Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T13:09:31Z","title":"A$^2$TGPO: Agentic Turn-Group Policy Optimization with Adaptive Turn-level Clipping","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-08T10:41:46.675257Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.06200"},"observation_digest":"sha256:b5055badad509b18bac471a333b5baf1b386ef9d7516fc9d4860f6c55e416ba4","observation_id":"ecd14ea2-58eb-438f-b41d-36cb35e4c4bb","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.06642","last_updated":"2026-05-07T17:51:16Z","snapshot_observed_at":"2026-07-06T23:19:05.764765Z","submitted_at":"2026-05-07T17:51:16Z","title":"StraTA: Incentivizing Agentic Reinforcement Learning with Strategic Trajectory Abstraction","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-08T09:59:48.604813Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.06642"},"observation_digest":"sha256:7fffb14e7970054989eefb1f785275a37b0a316c8522108caf2a1a8613ab40e3","observation_id":"4138b0f1-d58b-4d2e-90fd-a13ed32749f8","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.07276","last_updated":"2026-05-08T05:41:25Z","snapshot_observed_at":"2026-07-06T23:19:41.053744Z","submitted_at":"2026-05-08T05:41:25Z","title":"Signal Reshaping for GRPO in Weak-Feedback Agentic Code Repair","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-11T01:21:43.166304Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.07276"},"observation_digest":"sha256:b02d260f0ca4225a6b309441436078684138866b4fd89fa487e625edb57163b3","observation_id":"f50d1723-9254-49d8-906a-0436e958d562","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.08756","last_updated":"2026-05-09T07:36:45Z","snapshot_observed_at":"2026-07-06T23:20:57.084438Z","submitted_at":"2026-05-09T07:36:45Z","title":"AHD Agent: Agentic Reinforcement Learning for Automatic Heuristic Design","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T01:14:39.643357Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.08756"},"observation_digest":"sha256:13eb72da06091c5992d7911da4916a1ac6486c71df788970f63f048813104ba1","observation_id":"2764f7de-4e64-4785-b512-06afa90c7071","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.10064","last_updated":"2026-05-11T06:39:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-11T06:39:51Z","title":"MAGE: Multi-Agent Self-Evolution with Co-Evolutionary Knowledge Graphs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T05:13:28.089038Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.10064"},"observation_digest":"sha256:fd897d61c74aabdc9de870bb4220dc9708cf7dc15406265eaa5abb6c80763e13","observation_id":"c20f7f7b-8d3d-4f3d-ac8c-c47facd157b1","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.11853","last_updated":"2026-05-14T10:19:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-12T09:38:38Z","title":"GEAR: Granularity-Adaptive Advantage Reweighting for LLM Agents via Self-Distillation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T06:44:30.069444Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.11853"},"observation_digest":"sha256:460f4d5437be769e063fc43a1285eb315226e31cb288b6731bd97e3eeac10af1","observation_id":"3cfd89e0-f144-4799-a8e1-f62b5c3e7e6a","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.11853","last_updated":"2026-05-14T10:19:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-12T09:38:38Z","title":"GEAR: Granularity-Adaptive Advantage Reweighting for LLM Agents via Self-Distillation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-15T05:59:23.949124Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.11853"},"observation_digest":"sha256:f09057d4074fb263cb55f4bd3e54db5df2b3e9a0800e4d01425f6a69dcb79b57","observation_id":"9192f7c5-7de3-4233-b063-9daa01956329","resolution":{"observed_at":"2026-05-15T05:59:48.694854Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f8314963e516979090ce5fee3ea08293c8a022016f93cd5cb3b991fec7cfc6ac","observation_id":"7711b20f-5074-4cc0-b9e5-82086d47af10","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.14558","last_updated":"2026-05-14T08:33:02Z","snapshot_observed_at":"2026-07-06T23:25:58.895612Z","submitted_at":"2026-05-14T08:33:02Z","title":"Resolving Action Bottleneck: Agentic Reinforcement Learning Informed by Token-Level Energy","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-15T01:46:24.724553Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.14558"},"observation_digest":"sha256:e2ac84e4a0d1d77d450be71ed25ccbec1cb24098bbea3062e70acc4f16c81843","observation_id":"75fde33f-3688-4568-9140-b260bc16a4f5","resolution":{"observed_at":"2026-05-15T01:48:28.567125Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.16143","last_updated":"2026-05-15T16:24:16Z","snapshot_observed_at":"2026-07-06T23:27:20.429737Z","submitted_at":"2026-05-15T16:24:16Z","title":"Look Before You Leap: Autonomous Exploration for LLM Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T17:42:31.830822Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.16143"},"observation_digest":"sha256:cbc6e3ff0b3ce362d08729e6f4d1225cfd5881e386a3903046887d3111025de8","observation_id":"c69755d3-72f3-4d54-85b5-1e857cbae7c9","resolution":{"observed_at":"2026-05-20T17:43:36.378048Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.17486","last_updated":"2026-05-17T14:55:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-17T14:55:32Z","title":"DyGRO-VLA: Cross-Task Scaling of Vision-Language-Action Models via Dynamic Grouped Residual Optimization","version":1},"reference_index":144,"source":"arxiv_source","source_observed_at":"2026-05-20T12:39:50.004269Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.17486"},"observation_digest":"sha256:5e90ac5db0bb3187fb32f389ff160507d7be223e0938fed3b53a5960b8bb341d","observation_id":"75a2db13-0db5-450d-8454-1655c57c8a90","resolution":{"observed_at":"2026-05-20T12:43:17.165237Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.17792","last_updated":"2026-05-18T03:17:38Z","snapshot_observed_at":"2026-07-06T23:28:45.646975Z","submitted_at":"2026-05-18T03:17:38Z","title":"HydroAgent: Closing the Gap Between Frontier LLMs and Human Experts in Hydrologic Model Calibration via Simulator-Grounded RL","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-20T13:28:03.965754Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.17792"},"observation_digest":"sha256:e18b5e52c5b7806a54feecbf30bdd33b98037043d218c0c31689cedaf2c07988","observation_id":"a7e320ea-5775-48f6-8350-82f09467bb73","resolution":{"observed_at":"2026-05-20T13:28:18.828024Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.19447","last_updated":"2026-05-19T07:00:55Z","snapshot_observed_at":"2026-08-02T09:08:33.207894Z","submitted_at":"2026-05-19T07:00:55Z","title":"What and When to Distill: Selective Hindsight Distillation for Multi-Turn Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T05:41:23.712146Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.19447"},"observation_digest":"sha256:fa63ec9c1ab0a170d61631e7a07a1ee479129c32ac35330cc94abdd134b00b6f","observation_id":"92cb3926-1cd7-462c-acf9-e00d7658f33f","resolution":{"observed_at":"2026-05-20T05:43:05.690848Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.20061","last_updated":"2026-05-19T16:19:29Z","snapshot_observed_at":"2026-07-31T19:01:19.361800Z","submitted_at":"2026-05-19T16:19:29Z","title":"Rewarding Beliefs, Not Actions: Consistency-Guided Credit Assignment for Long-Horizon Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-20T05:35:45.084011Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.20061"},"observation_digest":"sha256:fff7f5ab307cddd9234a68d4158a971bc237cddf1e628f2eb879ad8e73599f87","observation_id":"8f6170ba-ec94-4679-9880-94e7e6a78c23","resolution":{"observed_at":"2026-05-20T05:38:05.463079Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.24426","last_updated":"2026-05-23T06:41:31Z","snapshot_observed_at":"2026-08-06T02:14:47.309550Z","submitted_at":"2026-05-23T06:41:31Z","title":"SEAL: Synergistic Co-Evolution of Agents and Learning Environments","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-30T13:38:10.466713Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.24426"},"observation_digest":"sha256:6f9aba42a62ff5f7812e288271099d213b59b9c3fc6a92702e092a00b6b3bc61","observation_id":"04ec941d-8ecb-4db9-9539-8cceb2e5f738","resolution":{"observed_at":"2026-06-30T13:44:40.937292Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.24828","last_updated":"2026-05-31T15:58:08Z","snapshot_observed_at":"2026-08-03T07:25:06.734747Z","submitted_at":"2026-05-24T02:41:44Z","title":"Test-Time Deep Thinking to Explore Implicit Rules","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-30T11:52:15.163893Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.24828"},"observation_digest":"sha256:d7d78646dbada2885ccbcc7f6a4c6ae7b118a6287c23622b9b9371d27961bb69","observation_id":"3b40b939-7bbe-4ec7-b924-ae2b29d3c4df","resolution":{"observed_at":"2026-06-30T11:54:38.222745Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.01811","last_updated":"2026-06-01T07:27:43Z","snapshot_observed_at":"2026-08-01T23:50:35.081089Z","submitted_at":"2026-06-01T07:27:43Z","title":"\"I've Seen How This Goes\": Characterizing Diversity via Progressive Conditional Surprise","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-28T14:51:06.448678Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.01811"},"observation_digest":"sha256:fd71480d5265fda9a8dc1c3594cc7283a047c9699d27b24d2887779da181a1c8","observation_id":"893c9511-cc09-442e-8e2c-52d730b24c45","resolution":{"observed_at":"2026-07-01T22:56:20.765294Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.02031","last_updated":"2026-06-04T09:03:30Z","snapshot_observed_at":"2026-08-03T02:24:10.083746Z","submitted_at":"2026-06-01T10:20:10Z","title":"OpenWebRL: Demystifying Online Multi-turn Reinforcement Learning for Visual Web Agents","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-28T15:46:50.587684Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.02031"},"observation_digest":"sha256:79a7cd30050093e8917969ad228cdd50c07ac844e5482c239017dba2aafb0425","observation_id":"30281e7c-4687-4ea5-a483-c266c1f14a32","resolution":{"observed_at":"2026-07-01T22:06:16.416446Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.02215","last_updated":"2026-06-01T13:16:22Z","snapshot_observed_at":"2026-08-02T18:02:20.147249Z","submitted_at":"2026-06-01T13:16:22Z","title":"Better with Experience: Self-Evolving LLM Agents for Evidence-Grounded Health Community Notes","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T14:20:13.991921Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.02215"},"observation_digest":"sha256:18b53f3841a677149f0cd5a4cb88758046bdead4b8a1094fee701311842913c1","observation_id":"09b01bac-cd26-4552-ad37-23c1c4b9fa62","resolution":{"observed_at":"2026-07-01T23:26:22.814833Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.02355","last_updated":"2026-06-01T15:02:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-01T15:02:59Z","title":"SIRI: Self-Internalizing Reinforcement Learning with Intrinsic Skills for LLM Agent Training","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-06-28T14:33:00.408984Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.02355"},"observation_digest":"sha256:9770543e1770f24b6574519be59509f6acbbcf37369c71f521972987c532d751","observation_id":"9995315c-f420-4a84-b9bd-b3cf5ecc40e3","resolution":{"observed_at":"2026-07-01T23:16:23.892126Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.02812","last_updated":"2026-06-01T19:30:07Z","snapshot_observed_at":"2026-07-06T23:43:07.940839Z","submitted_at":"2026-06-01T19:30:07Z","title":"Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-06-28T14:20:06.381334Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.02812"},"observation_digest":"sha256:f5a885d3baae7a158e3331e6bf43d6f9bed89ed260a0be554bd61eed668c5803","observation_id":"a4d95838-ab3e-4dc1-994b-19eb79d3053f","resolution":{"observed_at":"2026-06-28T14:22:17.935687Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.03108","last_updated":"2026-06-02T03:47:48Z","snapshot_observed_at":"2026-08-02T21:19:39.471703Z","submitted_at":"2026-06-02T03:47:48Z","title":"EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-06-28T10:30:42.057301Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.03108"},"observation_digest":"sha256:11e506c2bd22c6b483d3845ca59785ec23c2ec6b258877201aba1f866d5a66fe","observation_id":"14f83664-ee98-4aa8-bae0-4d0bc5d1a688","resolution":{"observed_at":"2026-07-02T02:56:29.172159Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.03762","last_updated":"2026-06-02T15:16:12Z","snapshot_observed_at":"2026-08-02T17:38:14.832178Z","submitted_at":"2026-06-02T15:16:12Z","title":"Tool-Aware Optimization with Entropy Guidance for Efficient Agentic Reinforcement Learning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T11:19:31.702516Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.03762"},"observation_digest":"sha256:d56dc202aac8fb2439497c914232f34ce92fef1eea0b55f24c7804f468b36fde","observation_id":"a7cf2cfc-85ae-4ad3-88ed-7106b7592131","resolution":{"observed_at":"2026-07-02T02:06:26.241307Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.04703","last_updated":"2026-06-03T10:30:09Z","snapshot_observed_at":"2026-07-06T23:44:47.848374Z","submitted_at":"2026-06-03T10:30:09Z","title":"Rethinking Continual Experience Internalization for Self-Evolving LLM Agents","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-28T06:29:11.398007Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.04703"},"observation_digest":"sha256:3fa856b08eeff36dd81a090cf1d63d12c2a13ed6a7c7cfe93507acb75c0b9d98","observation_id":"4206eb12-62b0-4082-b686-b923468fc54e","resolution":{"observed_at":"2026-07-02T07:56:47.799866Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.05885","last_updated":"2026-06-04T08:54:09Z","snapshot_observed_at":"2026-08-01T19:07:49.288895Z","submitted_at":"2026-06-04T08:54:09Z","title":"When Denser Credit Is Not Enough: Evidence-Calibrated Policy Optimization for Long-Horizon LLM Agent Training","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-28T02:17:32.324432Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.05885"},"observation_digest":"sha256:b97022beeade295496b41f2f46ee3bbfd3e9883e563be088c8a82403a5989ab7","observation_id":"cdc6b572-e970-4eb5-abe7-2e63e2b901dd","resolution":{"observed_at":"2026-07-02T12:16:56.809012Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.06708","last_updated":"2026-08-04T18:57:23Z","snapshot_observed_at":"2026-08-06T02:02:32.006926Z","submitted_at":"2026-06-04T20:48:37Z","title":"Signal-Driven Observation for Long-Horizon Web Agents","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-06-28T01:25:11.149977Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.06708"},"observation_digest":"sha256:a2d9c54f0ea44038d211a17d902d51aaa0862d7cf0ab651890756090915c8cbc","observation_id":"055d9650-f016-4f7e-b339-1092a20aeb76","resolution":{"observed_at":"2026-06-28T01:31:29.240542Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.09138","last_updated":"2026-06-08T07:35:18Z","snapshot_observed_at":"2026-07-06T23:48:30.569726Z","submitted_at":"2026-06-08T07:35:18Z","title":"Claw-R1: A Step-Level Data Middleware System for Agentic Reinforcement Learning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T17:28:58.574865Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.09138"},"observation_digest":"sha256:8fa1a9aa529495b31d244d72babdb9732ea3bcf1ac6e7a1b2498c85536a2ddf8","observation_id":"9216267e-9761-4ba9-affe-0bbfad21c897","resolution":{"observed_at":"2026-07-03T00:07:28.253101Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.09471","last_updated":"2026-06-08T13:28:54Z","snapshot_observed_at":"2026-07-06T23:48:48.962930Z","submitted_at":"2026-06-08T13:28:54Z","title":"Escaping the KL Agreement Trap in On-Policy Distillation","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-27T16:58:57.106367Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.09471"},"observation_digest":"sha256:608ce834720bf79225adb2ec2eff8848c77cbe1a2ebdf9b7c1b2ed6ca2930044","observation_id":"302ddb98-7377-47cd-bdea-3f923e19dafc","resolution":{"observed_at":"2026-07-03T00:57:29.474501Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.10507","last_updated":"2026-06-09T07:35:14Z","snapshot_observed_at":"2026-08-05T06:17:13.116592Z","submitted_at":"2026-06-09T07:35:14Z","title":"HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T13:04:28.758614Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.10507"},"observation_digest":"sha256:9c21c5ddfca49e79d0a872de731cfd3ba5bd43bba03165cb8e22926c3b4fa209","observation_id":"1535bad0-1d2e-4f17-9685-f9de617b5991","resolution":{"observed_at":"2026-07-03T05:47:41.643997Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.13262","last_updated":"2026-06-11T12:17:18Z","snapshot_observed_at":"2026-08-04T22:05:12.318613Z","submitted_at":"2026-06-11T12:17:18Z","title":"From Verdict to Process: Agentic Reinforcement Learning for Multi-Stage Fact Verification","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T06:47:09.681690Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.13262"},"observation_digest":"sha256:cfb47ca8a70e42c67cef882f8895146427548e5d0d64e6c3f01898fc0c56ea77","observation_id":"a415e6f4-e2c8-47ed-95f3-0ca19c67e792","resolution":{"observed_at":"2026-07-03T14:58:33.390469Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.17511","last_updated":"2026-06-16T04:42:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-16T04:42:43Z","title":"MagicSim: A Unified Infrastructure for Executable Embodied Interaction","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T01:00:07.465292Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.17511"},"observation_digest":"sha256:993c5bedc9c0703555ee4a4b67a8d9a320696018fb56da21762d96c8585c5500","observation_id":"89e55f14-1000-4b12-bcaa-986a3b883492","resolution":{"observed_at":"2026-07-03T20:58:58.092132Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.20475","last_updated":"2026-06-18T16:54:25Z","snapshot_observed_at":"2026-07-06T23:55:38.979306Z","submitted_at":"2026-06-18T16:54:25Z","title":"Marginal Advantage Accumulation for Memory-Driven Agent Self-Evolution","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-06-26T17:45:49.272070Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.20475"},"observation_digest":"sha256:6c5086ebca53786e26c33a0b39fb7884cbe5bcf59a8b7192fa79b7584360b1dd","observation_id":"4a7ab9b8-0470-45ee-a2e4-f46d6579b3a5","resolution":{"observed_at":"2026-07-04T03:39:30.791534Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.22995","last_updated":"2026-06-22T08:12:47Z","snapshot_observed_at":"2026-07-06T23:57:47.047975Z","submitted_at":"2026-06-22T08:12:47Z","title":"Group-Graph Policy Optimization for Long-Horizon Agentic Reinforcement Learning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-26T08:44:23.085858Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.22995"},"observation_digest":"sha256:dcc7d2122c50d6edd24b1d38b91dd1fa729e9692ceb74ee38a915e326033d725","observation_id":"20bee695-c39e-41b3-b1bf-92354649a765","resolution":{"observed_at":"2026-07-04T10:29:45.753940Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.23075","last_updated":"2026-06-22T09:23:50Z","snapshot_observed_at":"2026-08-02T06:16:18.387353Z","submitted_at":"2026-06-22T09:23:50Z","title":"Safety in Self-Evolving LLM Agent Systems: Threats, Amplification, and Case Studies","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-06-26T08:11:30.091859Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.23075"},"observation_digest":"sha256:84c151a14bfd42c74c212bb6cd6964b9828d825ba73bd76ac2f3a8452b10e925","observation_id":"4743eec0-3698-43dc-9ecd-e76ab5fefb42","resolution":{"observed_at":"2026-07-04T10:59:47.008995Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.26027","last_updated":"2026-06-24T16:55:56Z","snapshot_observed_at":"2026-08-05T06:50:33.242401Z","submitted_at":"2026-06-24T16:55:56Z","title":"Why Multi-Step Tool-Use Reinforcement Learning Collapses and How Supervisory Signals Fix It","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-25T19:28:36.499352Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.26027"},"observation_digest":"sha256:564bc060f35f0f47b55819b340b6d4576201fcc83d0eceff8c3fa877b1e8647a","observation_id":"f31d3551-2d7b-4670-9e87-00d8d3dcb77e","resolution":{"observed_at":"2026-07-04T20:50:12.373539Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2606.29502","last_updated":"2026-07-17T09:23:58Z","snapshot_observed_at":"2026-08-03T00:43:32.504636Z","submitted_at":"2026-06-28T17:02:18Z","title":"UCOB: Learning to Utilize and Evolve Agentic Skills via Credit-Aware On-Policy Bidirectional Self-Distillation","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-30T07:15:44.387347Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2606.29502"},"observation_digest":"sha256:cde9022317f3d884d4476188dc619f31c1365c1c26ea9002709970b61ca71601","observation_id":"5c01eefa-bda7-4e24-85e9-3336d39a822b","resolution":{"observed_at":"2026-06-30T07:24:22.387998Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-07-12T00:33:06.488657Z","title":"Information gain-based policy opti- mization: A simple and effective approach for multi-turn search agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03702","last_updated":"2026-07-04T04:45:05Z","snapshot_observed_at":"2026-08-03T14:40:44.889228Z","submitted_at":"2026-07-04T04:45:05Z","title":"Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-12T00:33:06.488657Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.03702"},"observation_digest":"sha256:7d6caa1c5e5cbe831c0d1aae17593d0cc51bbc7c49456b8fd02f4333f4ad0b57","observation_id":"42fbcc21-3a76-4e7c-9dc0-234b73e2a021","resolution":{"observed_at":"2026-07-12T00:33:06.488657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-07-11T20:42:48.585661Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning.arXiv preprint arXiv:2504.20073,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04242","last_updated":"2026-07-05T11:41:46Z","snapshot_observed_at":"2026-07-30T01:26:48.625826Z","submitted_at":"2026-07-05T11:41:46Z","title":"Progress- and Reliability-Oriented Group Policy Optimization for Agentic Reinforcement Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-11T20:42:48.585661Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.04242"},"observation_digest":"sha256:14532d585f9ceda78bbadf4f6396417dcbba844d7fc7deaf3961d134f0ab4582","observation_id":"81905e31-a397-4083-8ada-1d9aa26f3c55","resolution":{"observed_at":"2026-07-11T20:42:48.585661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-07-11T14:43:39.668059Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning.arXiv preprint arXiv:2504.20073, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04713","last_updated":"2026-07-06T06:32:39Z","snapshot_observed_at":"2026-08-05T03:26:47.178615Z","submitted_at":"2026-07-06T06:32:39Z","title":"RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-11T14:43:39.668059Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.04713"},"observation_digest":"sha256:78f95a9d026a0f7730f9f70b9d08e6fdb3d7b572d1071cb06d32dc1b05717d6f","observation_id":"a2f0c4dd-5b3b-4455-be53-92fa6e44ec70","resolution":{"observed_at":"2026-07-11T14:43:39.668059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2607.06140","last_updated":"2026-07-07T11:07:00Z","snapshot_observed_at":"2026-07-10T23:17:34.345638Z","submitted_at":"2026-07-07T11:07:00Z","title":"CurateEvo: Data-Curation Evolving for Agentic Post-Training","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-08T15:33:01.138366Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.06140"},"observation_digest":"sha256:3282ba04b3e9b18776a9bb2d37daac2713e1c6d208ff28d18a739e6081734c51","observation_id":"a49d9804-1e64-4d2b-b360-bef7c5926a61","resolution":{"observed_at":"2026-07-08T15:35:07.149976Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-07-14T12:26:27.446079Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.10350","last_updated":"2026-07-17T17:27:03Z","snapshot_observed_at":"2026-08-02T07:21:21.791047Z","submitted_at":"2026-07-11T15:24:43Z","title":"ABot-AgentOS: A General Robotic Agent OS with Lifelong Multi-modal Memory","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-07-14T12:26:27.446079Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.10350"},"observation_digest":"sha256:040abc42ae4f68e77066cc985d78c05e925bb776d5e0194fd3d48a36ab72e7bf","observation_id":"9366d44b-ed90-4935-8687-061518498d23","resolution":{"observed_at":"2026-07-14T12:26:27.446079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-02T07:21:32.935264Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.10350","last_updated":"2026-07-17T17:27:03Z","snapshot_observed_at":"2026-08-02T07:21:21.791047Z","submitted_at":"2026-07-11T15:24:43Z","title":"ABot-AgentOS: A General Robotic Agent OS with Lifelong Multi-modal Memory","version":3},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-02T07:21:32.935264Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.10350"},"observation_digest":"sha256:fab754d79734ddce216aae634e90bebb6cc7a3ad87a2752afc338be25fc1b750","observation_id":"4f621dae-7032-4da8-b946-fa806174ddbf","resolution":{"observed_at":"2026-08-02T07:21:32.935264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-07-14T10:33:54.851493Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning.arXiv preprint arXiv:2504.20073, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.10601","last_updated":"2026-07-12T06:38:55Z","snapshot_observed_at":"2026-07-16T23:18:52.250852Z","submitted_at":"2026-07-12T06:38:55Z","title":"Agentic-DPO: From Imitation to Agentic Policy Optimization on Expert Trajectories","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-14T10:33:54.851493Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.10601"},"observation_digest":"sha256:080a2e02f55d98428a17988f4e622dfd635e052b82f776e851fbe3ac03cfbe95","observation_id":"072b71ac-60e3-40ac-b0ad-f30414dc02fe","resolution":{"observed_at":"2026-07-14T10:33:54.851493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-02T04:59:08.940069Z","title":"arXiv preprint arXiv:2504.20073 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13539","last_updated":"2026-07-15T07:42:47Z","snapshot_observed_at":"2026-08-04T12:19:36.677375Z","submitted_at":"2026-07-15T07:42:47Z","title":"ThinkBLOX: 3D Indoor Scene Generation with Progressive Reasoning","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-02T04:59:08.940069Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.13539"},"observation_digest":"sha256:54a1751fcdcd9a3b895ebdad9a764ae903cf06e81aba36d6450c116d63c720ae","observation_id":"39265298-cdb6-48ae-8ec7-38b6236a36ec","resolution":{"observed_at":"2026-08-02T04:59:08.940069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-02T04:40:39.171769Z","title":"arXiv preprint arXiv:2504.20073 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14171","last_updated":"2026-07-15T09:37:36Z","snapshot_observed_at":"2026-08-05T13:44:31.832063Z","submitted_at":"2026-07-15T09:37:36Z","title":"Branching Policy Optimization: Sandbox-Native Language Agent Reinforcement Learning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T04:40:39.171769Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.14171"},"observation_digest":"sha256:7e6af0f41226164923a1675c9f4159488fe33d166c436e5e3d5858f66b59c977","observation_id":"ee512434-d41c-4c64-b9a2-d401a3209b85","resolution":{"observed_at":"2026-08-02T04:40:39.171769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-02T01:10:21.134716Z","title":"2504.20073 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14777","last_updated":"2026-07-16T09:57:18Z","snapshot_observed_at":"2026-08-05T21:32:12.968039Z","submitted_at":"2026-07-16T09:57:18Z","title":"SEED: Self-Evolving On-Policy Distillation for Agentic Reinforcement Learning","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-02T01:10:21.134716Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.14777"},"observation_digest":"sha256:3e23cf7a8895f3ca472545d28704b10c620d651022289566129c80484f35dd70","observation_id":"be013677-665f-491e-bb22-7ae8ab1adb42","resolution":{"observed_at":"2026-08-02T01:10:21.134716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-02T09:45:49.605488Z","title":"N., Liu, L., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16257","last_updated":"2026-06-28T08:35:47Z","snapshot_observed_at":"2026-08-02T09:45:48.181244Z","submitted_at":"2026-06-28T08:35:47Z","title":"From Outcomes to Actions: Leveraging Hindsight for Long-Horizon Language Agent Training","version":1},"reference_index":2008,"source":"pdf_text","source_observed_at":"2026-08-02T09:45:49.605488Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.16257"},"observation_digest":"sha256:268234efff38b2dc6338e51bbc961c4b1687bda4a7b681b4d3263754f2e4b65a","observation_id":"b6a23d0b-ff2c-4156-9c67-44fa9a8e2822","resolution":{"observed_at":"2026-08-02T09:45:49.605488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-01T21:04:42.348283Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16412","last_updated":"2026-07-17T18:04:38Z","snapshot_observed_at":"2026-08-02T16:27:43.759653Z","submitted_at":"2026-07-17T18:04:38Z","title":"Interactive Task Alignment as a POMDP","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-01T21:04:42.348283Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.16412"},"observation_digest":"sha256:bfeae0fd859e7e667cc3778a0e8c0331dd983cbe1940451ef4bbfba167d6394e","observation_id":"c31ec5f3-eb14-4508-9995-269e8aed0c6f","resolution":{"observed_at":"2026-08-01T21:04:42.348283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-02T08:56:36.117089Z","title":"N., Liu, L., Gottlieb, E., Lu, Y ., Cho, K., Wu, J., Fei-Fei, L., Wang, L., Choi, Y ., and Li, M","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19395","last_updated":"2026-07-03T05:07:22Z","snapshot_observed_at":"2026-08-03T03:43:09.098908Z","submitted_at":"2026-07-03T05:07:22Z","title":"From Trajectories to Prefixes: Reusing Teacher Trajectories via Replayed Prefixes and Online Continuation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T08:56:36.117089Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.19395"},"observation_digest":"sha256:c4bd0878dac5c92988860259691be7ca506324303ed02ca883d780792065c203","observation_id":"37782cd0-dcd7-4986-a590-a9cd9e2ec741","resolution":{"observed_at":"2026-08-02T08:56:36.117089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-01T10:42:41.407957Z","title":"arXiv preprint arXiv:2504.20073 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.20145","last_updated":"2026-07-30T09:30:44Z","snapshot_observed_at":"2026-08-03T18:47:24.558536Z","submitted_at":"2026-07-22T13:49:17Z","title":"SLAI T-Rex: Full-Parameter Post-training of the DeepSeek-V4 Family on Ascend SuperPOD","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-01T10:42:41.407957Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.20145"},"observation_digest":"sha256:8eaabc9c01c433dd2a9dd67effdec6bfcb2f6c282c5147002507d463b86b01b7","observation_id":"b7cbb21b-0bbc-453e-9235-562f0b29ee0c","resolution":{"observed_at":"2026-08-01T10:42:41.407957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-02T12:21:14.541478Z","title":"Tomer Wolfson, Mor Geva, Ankit Gupta, Matt Gard- ner, Yoav Goldberg, Daniel Deutch, and Jonathan Berant","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.20489","last_updated":"2026-06-04T14:21:19Z","snapshot_observed_at":"2026-08-02T12:21:12.913001Z","submitted_at":"2026-06-04T14:21:19Z","title":"EvoSQL: Memory-Augmented Critic-Generator Co-Evolution for Text-to-SQL","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T12:21:14.541478Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.20489"},"observation_digest":"sha256:87b476ade6052af2d44af7cc523671c12364698dc1b375c5b5ac508c0e18b0a8","observation_id":"a7f59ec8-ad52-45a0-86d2-be21b378a15f","resolution":{"observed_at":"2026-08-02T12:21:14.541478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-01T07:26:20.954574Z","title":"doi: 10.18653/v1/2024.acl-long.510.https://aclanthology.org/2024.acl-long.510/","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21461","last_updated":"2026-07-24T03:34:11Z","snapshot_observed_at":"2026-08-01T07:26:18.497566Z","submitted_at":"2026-07-23T16:05:46Z","title":"AREX: Towards a Recursively Self-Improving Agent for Deep Research","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T07:26:20.954574Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.21461"},"observation_digest":"sha256:435060134f746dfc070e9e936533c1c6e8f3559bdc71da39d5374ead81196e87","observation_id":"28f538ce-fa89-4896-a254-2995afafdb39","resolution":{"observed_at":"2026-08-01T07:26:20.954574Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-01T07:26:21.088294Z","title":"Jason Wei, Zhiqing Sun, Spencer Papay, Scott McKinney, Jeffrey Han, Isa Fulford, Hyung Won Chung, Alex Tachard Passos, William Fedus, and Amelia Glaese","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21461","last_updated":"2026-07-24T03:34:11Z","snapshot_observed_at":"2026-08-01T07:26:18.497566Z","submitted_at":"2026-07-23T16:05:46Z","title":"AREX: Towards a Recursively Self-Improving Agent for Deep Research","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T07:26:21.088294Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2607.21461"},"observation_digest":"sha256:72ff6b53831bf9b1f10a88873aa45df8855a64bb8030a8f0dc39b9a01f2158e1","observation_id":"424ee099-0126-4d97-9e0b-a3d9bc69968b","resolution":{"observed_at":"2026-08-01T07:26:21.088294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2504.20073/citation-record","integrity":"/paper/2504.20073/integrity","json":"/paper/2504.20073/citation-record.json","paper":"/paper/2504.20073"},"outbound":[],"paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 100 inbound Pith citation observations for arXiv:2504.20073."}