{"as_of":"2026-08-06T18:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a2bb8a38b8aa96ab240cab746084cd9f4795bc19ce7f2ebde2ba72bf85affecd","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-03T18:47:46.719344Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T09:49:31.023088Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.01120","snapshot_observed_at":"2026-08-01T09:49:31.023088Z","title":"Next-generation agentic reinforcement learning systems enable self-evolving agents","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.21653","last_updated":"2026-07-22T18:06:15Z","snapshot_observed_at":"2026-08-06T08:38:08.188489Z","submitted_at":"2026-07-22T18:06:15Z","title":"Molt: A Scalable PyTorch-Native Training Framework for Agentic Reinforcement Learning","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-01T09:49:31.023088Z"},"links":{"cited_paper":"/paper/2607.01120","citing_paper":"/paper/2607.21653"},"observation_digest":"sha256:0ac9eef7eda17fdc63dc842380213419e10c937dd1df0a29c62dc213f85247fa","observation_id":"6a869fb7-4c00-430a-973c-afe7325b5047","resolution":{"observed_at":"2026-08-01T09:49:31.023088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2607.01120/citation-record","integrity":"/paper/2607.01120/integrity","json":"/paper/2607.01120/citation-record.json","paper":"/paper/2607.01120"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.766579Z","title":"Openclaw: The ai that actually does things, 2026","venue":null,"work_id":"d1326464-42e6-44e5-9e64-5392a471f119","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:7bc856a6c9cc4543678fa3fa7df396263b2928f51242d78e478a456dc62480b4","observation_id":"e1c90530-8511-499e-a8b0-d26b7a6805f1","resolution":{"observed_at":"2026-07-05T03:30:39.196295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.10165","last_updated":"2026-05-11T10:03:41Z","snapshot_observed_at":"2026-07-06T22:48:35.345421Z","submitted_at":"2026-03-10T18:59:01Z","title":"OpenClaw-RL: Train Any Agent Simply by Talking","version":2},"cited_work":{"arxiv_id":"2603.10165","doi":"10.48550/arxiv.2603.10165","metadata_source":"pith","pith_arxiv_id":"2603.10165","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenClaw-RL: Train Any Agent Simply by Talking","venue":"cs.CL","work_id":"78607317-8305-4515-8dc3-20b4ff5b8f3a","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2603.10165","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:c2eb2936f206cd211c7b65616a3f060ac6b98454d75a4d9875c55fb9d2d790f3","observation_id":"1b3b92dd-efc9-48a8-b07c-4b4fe1ddfee2","resolution":{"observed_at":"2026-07-03T18:48:49.543357Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.17187","doi":"10.48550/arxiv.2603.17187","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Metaclaw: Just talk–an agent that meta-learns and evolves in the wild","venue":"arXiv (Cornell University)","work_id":"118635c4-eee7-48da-9dfe-ffaf07704aa7","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:05a95f33ef193ccc3a5bb747263431aaf521f0780cc89132b2bf826f2e3ea181","observation_id":"967125cd-f96c-4e9c-b929-22c36418c295","resolution":{"observed_at":"2026-07-03T18:48:49.564921Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.08377","last_updated":"2026-04-09T15:38:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-09T15:38:27Z","title":"SkillClaw: Let Skills Evolve Collectively with Agentic Evolver","version":1},"cited_work":{"arxiv_id":"2604.08377","doi":"10.48550/arxiv.2604.08377","metadata_source":"pith","pith_arxiv_id":"2604.08377","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"SkillClaw: Let Skills Evolve Collectively with Agentic Evolver","venue":"cs.AI","work_id":"31a44ebb-9aae-4719-8d62-50f8c82ece16","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2604.08377","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:56c56a9c03e0cf57939d0b22389f35c2cc67b25d162a5ca90f95b91af937f879","observation_id":"04c9d290-5979-48eb-9048-d581db4657d6","resolution":{"observed_at":"2026-07-03T18:48:49.546339Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.760708Z","title":"Reflexion: Language agents with verbal reinforcement learning","venue":null,"work_id":"500033db-8592-4c12-b18b-3eff8505dd06","year":2023},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:0dd338ec2016a7ca72ad2574e3d110b810a532886961584cf2f2e95fd227552b","observation_id":"2fe8fb89-55af-4473-881c-7459d52f566f","resolution":{"observed_at":"2026-07-05T03:30:39.206472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.18743","doi":"10.48550/arxiv.2603.18743","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Memento-skills: Let agents design agents","venue":"arXiv (Cornell University)","work_id":"53800d8e-8b59-418c-b6cb-3cd56186c42e","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:55d77d6a08e8bc2e2ed7bd51560b595a10afacd0a76efac276b09ad8cf68148b","observation_id":"d6e51a11-f1ef-438b-b3f4-2e2e0894cdd5","resolution":{"observed_at":"2026-07-03T18:48:49.535633Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.757635Z","title":"Agentic context engineering: Evolving contexts for self-improving language models","venue":null,"work_id":"ccda3996-ef88-4e9e-85a9-615f06e3c913","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:b1fbedfcfff710a0c51486868cd64fb3ea967303f702df2d21c91d5996ae55e5","observation_id":"ae361a64-5409-4815-89ff-015f892e6109","resolution":{"observed_at":"2026-07-05T03:30:39.208651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.788791Z","title":"Areal: A large-scale asynchronous reinforcement learning system for language reasoning","venue":null,"work_id":"5aaf9ce4-ea52-4b39-b410-b6dd8069fdb1","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:11efbd5a699a41b81c0fffe0eccdb1beff8f657a3486c2efbc6056df62e7f246","observation_id":"dca7b9d2-8929-46f6-8200-13b76856f345","resolution":{"observed_at":"2026-07-05T03:30:39.208460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"cited_work":{"arxiv_id":"2509.08827","doi":"10.48550/arxiv.2509.08827","metadata_source":"pith","pith_arxiv_id":"2509.08827","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","venue":"cs.CL","work_id":"7618c14b-e527-4268-9926-d4f462ea9925","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2509.08827","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:c2b06bf8130d9346f67a240981305f62e387ff218b2757fe1aad1696d4b3426f","observation_id":"b1e742bc-aef7-445d-8132-a304f92dcdbd","resolution":{"observed_at":"2026-07-03T18:48:49.532549Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:04:52.382298Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"2dc9c508-76e0-4e0e-b3ca-cc5656b43a68","year":2022},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:009b5f4bb4febec147c6db8782741e670de7146cbf2b7bddf01c3b90ac21b61a","observation_id":"efe2340e-a09f-4d24-ac53-05179d597600","resolution":{"observed_at":"2026-07-05T03:30:39.218276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":"2212.08073","doi":"10.48550/arxiv.2212.08073","metadata_source":"pith","pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Constitutional AI: Harmlessness from AI Feedback","venue":"cs.CL","work_id":"faaaa4e0-2676-4fac-a0b4-99aef10d2095","year":2022},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:80e22a55e034624e3f561b6ca53fc0c980dfde5c6609e549dc87eeaecaf18700","observation_id":"8df964e6-5775-4b93-897b-8a6b9c73b844","resolution":{"observed_at":"2026-07-03T18:48:49.527409Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.114855+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.114855+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:77b55eb40a4ab56f76fd3bc4796ce9cdfbe7ced3f3099027ad0b7393db428e28","observation_id":"96ec482e-174a-4b88-b947-04ac42f8395d","resolution":{"observed_at":"2026-07-03T18:48:49.529749Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T22:47:43.126704Z","title":"Direct preference optimization: Your language model is secretly a reward model.Advances in neural information processing systems, 36:53728–53741","venue":null,"work_id":"d65a86ed-cf86-475e-9502-161e34d6a682","year":2023},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:6ed110723362754e828cf3e4da2749d51c102c80b87c5573064f4b9075c02595","observation_id":"06eec497-5edc-48f7-81a0-5663b9d3136a","resolution":{"observed_at":"2026-07-05T03:30:39.210686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:9d0496f6e861ca2a85d5012e436ed70cd136b23864435b5b1b1571df569ddd17","observation_id":"b7e9d4cf-9adc-4873-9c94-9cb7193c9153","resolution":{"observed_at":"2026-07-03T18:48:49.540942Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02547","last_updated":"2026-04-17T18:09:08Z","snapshot_observed_at":"2026-08-03T09:07:42.489237Z","submitted_at":"2025-09-02T17:46:26Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","version":5},"cited_work":{"arxiv_id":"2509.02547","doi":"10.48550/arxiv.2509.02547","metadata_source":"pith","pith_arxiv_id":"2509.02547","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","venue":"cs.AI","work_id":"87909127-da20-4ccc-8ae3-4a4a20ef81b7","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2509.02547","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:71f8be67db0fef19dad1d89dbd01908384aa44be35bcec3d43e56f8fd22884ef","observation_id":"76af6b6d-0be3-4282-b302-959eb43ce550","resolution":{"observed_at":"2026-07-03T18:48:49.516258Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":"2210.03629","doi":"10.48550/arxiv.2210.03629","metadata_source":"pith","pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","venue":"cs.CL","work_id":"407a2351-25f1-497d-b611-f77d0292a8e6","year":2022},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:607d8bbeba453f1320a86f285c08643d4f379da1b829b12e84da6511cfc890ba","observation_id":"4b4ace6b-a685-4d6c-878a-839beb201afc","resolution":{"observed_at":"2026-07-03T18:48:49.519138Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:36.897515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:36.897515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07496","last_updated":"2024-06-11T17:32:21Z","snapshot_observed_at":"2026-07-06T18:29:04.294349Z","submitted_at":"2024-06-11T17:32:21Z","title":"TextGrad: Automatic \"Differentiation\" via Text","version":1},"cited_work":{"arxiv_id":"2406.07496","doi":"10.48550/arxiv.2406.07496","metadata_source":"pith","pith_arxiv_id":"2406.07496","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"TextGrad: Automatic \"Differentiation\" via Text","venue":"cs.CL","work_id":"c52b4841-a4b4-4b4e-bc86-8ef6645a9bbb","year":2024},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2406.07496","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:41ca95f595f605c078a80b860dd9df65539c1a16bedbbfbf8be9292599a52cc7","observation_id":"e6c8c4c1-7b6f-4ead-a109-8ba3a0fb3888","resolution":{"observed_at":"2026-07-03T18:48:49.521976Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:25.209279+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:25.209279+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.773940Z","title":"Unlocking long-horizon agentic search with large-scale end-to-end rl","venue":null,"work_id":"96f2749a-3ebc-4a26-875d-905956815054","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:b36a610f7a43f5bee65d2747a90f7a81b27d30d116e7372a9944940b3b7092cf","observation_id":"7311449a-4606-4d17-8358-8fec45566b0d","resolution":{"observed_at":"2026-07-05T03:30:39.187826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14088","last_updated":"2025-04-24T13:24:42Z","snapshot_observed_at":"2026-07-06T18:34:07.264237Z","submitted_at":"2024-06-20T08:04:07Z","title":"ReaL: Efficient RLHF Training of Large Language Models with Parameter Reallocation","version":2},"cited_work":{"arxiv_id":"2406.14088","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14088","snapshot_observed_at":"2026-07-03T18:48:49.536779Z","title":"Realhf: Optimized rlhf training for large language models through parameter reallocation","venue":null,"work_id":"ef22908e-c279-4d33-a979-6ae77585a9bd","year":2024},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2406.14088","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:e8eb1615fdef8246f6cb56bab91bbbc3dc8ae83980b3d6619f6926471f1f730f","observation_id":"06870b1d-8722-4fea-8bb3-1141e75b3305","resolution":{"observed_at":"2026-07-03T18:48:49.538132Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.747605Z","title":"Optimizing {RLHF} training for large language models with stage fusion","venue":null,"work_id":"6b83eeaa-040c-46a0-830e-3bf49183ad2d","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:d65e6c36ffd7b8adc817107297ab59866cb9bfecabf5735351ce5b204bbc42d1","observation_id":"7dcf6429-605c-42ca-925d-cafe84bc77f9","resolution":{"observed_at":"2026-07-05T03:30:39.204090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.22789","last_updated":"2025-07-31T02:18:13Z","snapshot_observed_at":"2026-07-06T22:05:12.838227Z","submitted_at":"2025-07-30T15:55:08Z","title":"G-Core: A Simple, Scalable and Balanced RLHF Trainer","version":2},"cited_work":{"arxiv_id":"2507.22789","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.22789","snapshot_observed_at":"2026-07-03T18:48:49.523099Z","title":"G-core: A simple, scalable and balanced rlhf trainer","venue":null,"work_id":"55fa79c4-eeb0-4fa5-abdf-03a2b4e06d86","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2507.22789","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:f145433966e24cfe576b5407200064aeeee5d75e3d1bdb16beb33422f70098e1","observation_id":"9487bfb8-2fb1-45b3-a0a8-28adfd9e3a54","resolution":{"observed_at":"2026-07-03T18:48:49.524813Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T17:35:10.023423Z","title":"Hybridflow: A flexible and efficient rlhf framework","venue":null,"work_id":"37ada09c-055c-4ab6-99e5-393107343cd6","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:e0058344df68aad16a4fc6081a8831f252370298ebfdc58c05e46b74da40220a","observation_id":"79e78776-42d1-406d-8f8c-808815f3e29f","resolution":{"observed_at":"2026-07-05T03:30:39.212609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15930","last_updated":"2025-04-22T14:19:06Z","snapshot_observed_at":"2026-07-06T21:13:04.549829Z","submitted_at":"2025-04-22T14:19:06Z","title":"StreamRL: Scalable, Heterogeneous, and Elastic RL for LLMs with Disaggregated Stream Generation","version":1},"cited_work":{"arxiv_id":"2504.15930","doi":"10.48550/arxiv.2504.15930","metadata_source":"arxiv_reference","pith_arxiv_id":"2504.15930","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamrl: Scalable, heterogeneous, and elastic rl for llms with disaggregated stream generation","venue":"ArXiv.org","work_id":"63df3644-702d-4ae6-8e4a-954a41887545","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2504.15930","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:be36e8fd23e877781b008e187b348f93f261902299b105b3b0b1eb64b2d53dda","observation_id":"082f5c95-f61d-4dd9-8c06-0462846c29b6","resolution":{"observed_at":"2026-07-03T18:48:49.509444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-13T08:50:16.320291+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T08:50:16.320291+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01663","last_updated":"2025-07-02T12:45:34Z","snapshot_observed_at":"2026-07-06T21:51:04.907326Z","submitted_at":"2025-07-02T12:45:34Z","title":"AsyncFlow: An Asynchronous Streaming RL Framework for Efficient LLM Post-Training","version":1},"cited_work":{"arxiv_id":"2507.01663","doi":"10.48550/arxiv.2507.01663","metadata_source":"arxiv_reference","pith_arxiv_id":"2507.01663","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Asyncflow: An asynchronous streaming rl framework for efficient llm post-training","venue":"ArXiv.org","work_id":"c78dd69f-0c92-450b-8e2b-dec1ded07146","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2507.01663","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:b943fe56fd66e722dd966be3028f74c61c08fc7534fcd09b5b58a187ef62ce51","observation_id":"17173bbf-93a2-485d-9c69-c7d3c3888201","resolution":{"observed_at":"2026-07-03T18:48:49.549581Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.781446Z","title":"Introducing the Model Context Protocol","venue":null,"work_id":"510486d0-c9d9-4bec-8919-72dcc4577fbb","year":2024},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:ea5fc6c25ba095e1487e46017642fc60c04e2452c78eed4170429dcda8278a0f","observation_id":"7c103216-7fca-42f0-8da6-95001fddfca1","resolution":{"observed_at":"2026-07-05T03:30:39.197808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.771453Z","title":"Agent2agent (a2a) protocol.https://a2a-protocol.org/, 2025","venue":null,"work_id":"279f7614-551a-4acb-821f-67a29595f26d","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:3b5d5cd087ef98649914efa22a506efce3791debf8a6e4c96b6d011bc772e982","observation_id":"07fc8c61-7a8c-402a-a713-6ed78e051e15","resolution":{"observed_at":"2026-07-05T03:30:39.194269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16736","last_updated":"2025-06-21T17:39:43Z","snapshot_observed_at":"2026-07-31T15:41:38.725206Z","submitted_at":"2025-04-23T14:07:26Z","title":"A Survey of AI Agent Protocols","version":3},"cited_work":{"arxiv_id":"2504.16736","doi":"10.48550/arxiv.2504.16736","metadata_source":"pith","pith_arxiv_id":"2504.16736","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey of AI Agent Protocols","venue":"cs.AI","work_id":"7233e34f-dc80-449b-94df-ceb2ee783ea5","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2504.16736","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:9030c8dd8b4b4adc84e1210af0aca569ab919102a672468936fa486c31726645","observation_id":"79ec63d6-1e71-4b4b-b3e7-b1c6f0664c4b","resolution":{"observed_at":"2026-07-03T18:48:49.556107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.07219","last_updated":"2021-02-06T01:57:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-04-15T17:18:19Z","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","version":4},"cited_work":{"arxiv_id":"2004.07219","doi":"10.48550/arxiv.2004.07219","metadata_source":"pith","pith_arxiv_id":"2004.07219","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","venue":"cs.LG","work_id":"47082e4e-a4a5-418b-bf4f-4667355065fc","year":2020},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2004.07219","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:221db249f2cab3ad8e610a7976de10738e7f4fff53db776d853c85f2dcbe9012","observation_id":"aad236f2-dc72-4606-960e-cccab34bc2f6","resolution":{"observed_at":"2026-07-03T18:48:49.553561Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02767","last_updated":"2021-11-04T11:48:19Z","snapshot_observed_at":"2026-07-06T12:05:22.076764Z","submitted_at":"2021-11-04T11:48:19Z","title":"RLDS: an Ecosystem to Generate, Share and Use Datasets in Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2111.02767","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.02767","snapshot_observed_at":"2026-07-03T18:48:49.557451Z","title":"Rlds: an ecosystem to generate, share and use datasets in reinforcement learning","venue":null,"work_id":"a97714ba-5cd9-4d3a-aac5-c9ede79d87d3","year":2021},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"cited_paper":"/paper/2111.02767","citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:8790f002179199603703ab5453fe8b904c61f12ce0c03f7612b6f7b9b7e896b4","observation_id":"9092a0e7-8054-4066-95f6-fc2fc0a351e6","resolution":{"observed_at":"2026-07-03T18:48:49.559274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.24702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T18:48:49.560270Z","title":"Agent data protocol: Unifying datasets for diverse, effective fine-tuning of llm agents","venue":null,"work_id":"6afdfaca-1d33-44b7-97e7-298e23f6c60c","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:52ccf00a82815bdca1efb7f99a9b80def55ad035884c3427496a5e2e234f4f6f","observation_id":"3aa93142-cd4f-4de2-8eb6-38292f3c8ac6","resolution":{"observed_at":"2026-07-03T18:48:49.561726Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.779054Z","title":"LangChain: The agent engineering platform","venue":null,"work_id":"24f13792-9500-42f4-aebe-9e9631bcef47","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:8b224a0a22bc53b82c92be700f86481f1af140faa3fc8c78ca8a0814b59a41aa","observation_id":"c8bf950d-8111-425e-9de1-2afa6e8be718","resolution":{"observed_at":"2026-07-05T03:30:39.206668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.750934Z","title":"LangGraph: Build resilient language agents as graphs","venue":null,"work_id":"3a1b52b5-3b60-47ee-8315-87fd081513e3","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:bb8b473cdc9eae537748f56b817ecb8c62db320028f9527a860b65d238507f96","observation_id":"2db5f868-cab5-4c31-865f-3fed96a8be40","resolution":{"observed_at":"2026-07-05T03:30:39.216644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.741941Z","title":"CrewAI: Framework for orchestrating role-playing, autonomous AI agents","venue":null,"work_id":"f214ca5b-383d-423d-8bbd-7085dde130c9","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:64d2fca2856da889d76cfd6228716d07bd86b5bc31e8f7c6b22adbc68a8dbb6f","observation_id":"41e9af60-311a-4abb-8ce8-1c387b0e3477","resolution":{"observed_at":"2026-07-05T03:30:39.191761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.791404Z","title":"OpenAI Agents SDK: A lightweight, powerful framework for multi-agent workflows","venue":null,"work_id":"76863cfe-e81d-4c28-94d1-e734c2ffe4cc","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:f1940e9062d7cca69b2d01298888a9dc71e5f42312459d447714b7adf46f8343","observation_id":"eeafd810-b69f-4294-8a62-ab7eb4ff6b8c","resolution":{"observed_at":"2026-07-05T03:30:39.201810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.768968Z","title":"Claude Agent SDK.https://github.com/anthropics/claude-agent-sdk-python, 2025","venue":null,"work_id":"32abe964-001d-4246-b661-4e9006987434","year":2025},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:2ee49a4307a4dc8da64209252056f3122a6b75cefca43e23a02fb479fea5c4c2","observation_id":"639b0062-83c4-4c49-a225-30d408f222ee","resolution":{"observed_at":"2026-07-05T03:30:39.210562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.763425Z","title":"Agentprm: Process reward models for llm agents via step-wise promise and progress","venue":null,"work_id":"b22afe0f-08ed-410f-8931-e9e506be9dfe","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:7ddc5249dda329d16895b226829bbdda34bebe4429da0d0024286a3c83f5d9ab","observation_id":"73883c96-3a86-45eb-b16f-3cc6483ff40b","resolution":{"observed_at":"2026-07-05T03:30:39.214498Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.02488","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T18:48:49.510756Z","title":"Rlanything: Forge environment, policy, and reward model in completely dynamic rl system","venue":null,"work_id":"6d0116cf-2531-41e0-a250-af72f29c9448","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:a1da51d63418cb0ea99b2e6da9d54ce6ba78fc5466311a1654e65a96ca7db3a9","observation_id":"e4742587-7e2f-4d8b-99c3-607576e104d8","resolution":{"observed_at":"2026-07-03T18:48:49.512855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:52:03.786373Z","title":"Hermes agent: The self-improving ai agent built by nous research","venue":null,"work_id":"4c9b79ce-974c-4fe5-8e34-6f1699f6996a","year":2026},"citing_paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-03T18:47:46.719344Z"},"links":{"citing_paper":"/paper/2607.01120"},"observation_digest":"sha256:99f20d392998811cb6b83fd29d06b9eb52087ab49e0b104e4c4995a8ee4f3842","observation_id":"5e8abccd-d426-4fc2-8f39-021a16556997","resolution":{"observed_at":"2026-07-05T03:30:39.216467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.01120","last_updated":"2026-07-02T14:02:28Z","latest_version":2,"primary_category":"cs.DC","snapshot_observed_at":"2026-07-07T00:06:44.379624Z","submitted_at":"2026-07-01T16:08:02Z","title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":20,"verified_fuzzy":18},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 1 inbound Pith citation observation for arXiv:2607.01120."}