{"as_of":"2026-08-06T10:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7a392edbfa31c9f38bf7eb97dc0b82bd23b5a1b8fb1ac71a250771317edb6568","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T04:28:49.464315Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.22529/citation-record","integrity":"/paper/2607.22529/integrity","json":"/paper/2607.22529/citation-record.json","paper":"/paper/2607.22529"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:44.927193Z","title":"Ultraif: Advancing instruction following from the wild","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:44.927193Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:bc8fee1c3b44a8f521c389c9941a0387784f89b4fce07438432d6b0dec545a2d","observation_id":"0aa6daad-b7c5-4289-8cdf-61e05fa1dbdc","resolution":{"observed_at":"2026-08-01T04:28:44.927193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.11543","last_updated":"2026-06-10T01:11:50Z","snapshot_observed_at":"2026-07-06T23:50:35.052764Z","submitted_at":"2026-06-10T01:11:50Z","title":"SkillJuror: Measuring How Agent Skill Organization Changes Runtime Behavior","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.11543","snapshot_observed_at":"2026-08-01T04:28:45.144472Z","title":"Skilljuror: Measuring how agent skill organization changes runtime behavior.arXiv preprint arXiv:2606.11543, 2026b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.144472Z"},"links":{"cited_paper":"/paper/2606.11543","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:077391e32fe5695248d4882212440737a461e8ef516db4343063fef4500323f1","observation_id":"a535cdee-e078-49be-b9bc-64cd775d6504","resolution":{"observed_at":"2026-08-01T04:28:45.144472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:49.404490Z","title":"I want to plan a dinner for my friends this weekend in Portland. We’re looking for a nice Italian place that doesn’t cost more than $30 per person","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.404490Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:83a1c859ee65886c14d91d39a81715f3db2e785c9bc81949d6824360d174f2e8","observation_id":"c82e338f-656d-468f-ac41-3df830a9e4b3","resolution":{"observed_at":"2026-08-01T04:28:49.404490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:45.458502Z","title":"From self-evolving synthetic data to verifiable-reward rl: Post-training multi-turn interactive tool-using agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.458502Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:645697f7d0a801465bfbd8bb27612e6caed837dffa32aa51f0cc329c9bddb836","observation_id":"3b8b6ba3-07ec-41c1-8899-32f9e7845c05","resolution":{"observed_at":"2026-08-01T04:28:45.458502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02089","last_updated":"2025-02-18T11:39:46Z","snapshot_observed_at":"2026-07-06T19:26:38.002071Z","submitted_at":"2024-10-02T23:25:17Z","title":"RLEF: Grounding Code LLMs in Execution Feedback with Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02089","snapshot_observed_at":"2026-08-01T04:28:45.552110Z","title":"Rlef: Grounding code llms in execution feedback with reinforcement learning.arXiv preprint arXiv:2410.02089,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.552110Z"},"links":{"cited_paper":"/paper/2410.02089","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:9ef94627e9b1131807afe5ccf15d2f647eae0c06dd29a58774d331094a247ce3","observation_id":"bba3d182-65cb-4f0c-a6a5-429435763b84","resolution":{"observed_at":"2026-08-01T04:28:45.552110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:45.639596Z","title":"Gems: Agent-native multimodal generation with memory and skills.arXiv preprint arXiv:2603.28088,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.639596Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:86a7e75d4bf2a8809d5664ca770ec2a299604ffdac98f09508e72e3bb7c9883d","observation_id":"2efc86d2-a72e-4ffe-a41c-9dce29985975","resolution":{"observed_at":"2026-08-01T04:28:45.639596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.05004","last_updated":"2026-02-13T18:53:32Z","snapshot_observed_at":"2026-07-06T22:09:08.864232Z","submitted_at":"2025-08-07T03:38:16Z","title":"R-Zero: Self-Evolving Reasoning LLM from Zero Data","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.05004","snapshot_observed_at":"2026-08-01T04:28:45.732564Z","title":"R-zero: Self-evolving reasoning llm from zero data.arXiv preprint arXiv:2508.05004, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.732564Z"},"links":{"cited_paper":"/paper/2508.05004","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:cd13ade0eb154fcabf97ed3ea2d733431f20cf558a807156efa20a8fe5dae87a","observation_id":"d0f755cd-4170-43dd-ac7a-4442881dbc90","resolution":{"observed_at":"2026-08-01T04:28:45.732564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:45.841654Z","title":"Bowen Jiang, Taiwei Shi, Ryo Kamoi, Yuan Yuan, Camillo J Taylor, Longqi Yang, Pei Zhou, and Sihao Chen","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.841654Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:9145c5a341d8e59058e2038b6595776d2f88307f08164f796a3cd28bc9603131","observation_id":"1187567d-5631-46c9-8bae-981a4ff62159","resolution":{"observed_at":"2026-08-01T04:28:45.841654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03714","last_updated":"2023-10-05T17:37:25Z","snapshot_observed_at":"2026-08-04T05:21:05.846165Z","submitted_at":"2023-10-05T17:37:25Z","title":"DSPy: Compiling Declarative Language Model Calls into Self-Improving Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03714","snapshot_observed_at":"2026-08-01T04:28:45.954722Z","title":"Omar Khattab, Arnav Singhvi, Paridhi Maheshwari, Zhiyuan Zhang, Keshav Santhanam, Sri Vardhamanan, Saiful Haq, Ashutosh Sharma, Thomas T Joshi, Hanna Moazam, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.954722Z"},"links":{"cited_paper":"/paper/2310.03714","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:0083cbcdc4c84f53f90ebcf796d4fd2c769c3d39af1910e8d683bee62630d794","observation_id":"64db138e-10e3-4c85-9ef0-86e5ca864eb2","resolution":{"observed_at":"2026-08-01T04:28:45.954722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:46.184528Z","title":"R-diverse: Mitigating diversity illusion in self-play llm training.arXiv preprint arXiv:2602.13103, 2026a","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.184528Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:f07662f248ccd1b918623df81f434dd83268ff1cf04b638bed6ae83ca15d68b4","observation_id":"434e22b1-81ff-4212-a799-f31a0f7c68af","resolution":{"observed_at":"2026-08-01T04:28:46.184528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12670","last_updated":"2026-03-13T07:33:01Z","snapshot_observed_at":"2026-07-06T22:45:44.135338Z","submitted_at":"2026-02-13T07:06:06Z","title":"SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.12670","snapshot_observed_at":"2026-08-01T04:28:46.307495Z","title":"Skillsbench: Benchmarking how well agent skills work across diverse tasks.arXiv preprint arXiv:2602.12670, 2026b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.307495Z"},"links":{"cited_paper":"/paper/2602.12670","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:0b4de8caa5ad8ddcba0feb9b8afb2235bc9c37c23d0cc2b08c6c284ef1e23ef8","observation_id":"bde93b8d-d7c9-49f3-b762-451245bfa63e","resolution":{"observed_at":"2026-08-01T04:28:46.307495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:46.439202Z","title":"Agent skills: A data-driven analysis of claude skills for extending large language model functionality.arXiv preprint arXiv:2602.08004,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.439202Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:ae99893e90c0a8755de2ff3cfe0569e3551e436ba0efbe571c013799f3503c6c","observation_id":"e96f8b1d-10c6-4b08-bdf5-ad7be70090cf","resolution":{"observed_at":"2026-08-01T04:28:46.439202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.08584","last_updated":"2026-01-13T14:06:03Z","snapshot_observed_at":"2026-08-04T17:10:53.354037Z","submitted_at":"2026-01-13T14:06:03Z","title":"Ministral 3","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.08584","snapshot_observed_at":"2026-08-01T04:28:46.545451Z","title":"Ministral 3.arXiv preprint arXiv:2601.08584, 2026a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.545451Z"},"links":{"cited_paper":"/paper/2601.08584","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:dbb63bdfc75ef781f7bacee6a78d6296fb61bc2faf910be6ce8fec4f02afc5d6","observation_id":"0f867384-7fb6-4afc-a086-205fb99b4c50","resolution":{"observed_at":"2026-08-01T04:28:46.545451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.01139","last_updated":"2026-06-17T05:10:07Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T10:19:13Z","title":"SkillRevise: Improving LLM-Authored Agent Skills via Trace-Conditioned Skill Revision","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.01139","snapshot_observed_at":"2026-08-01T04:28:46.657137Z","title":"Skillrevise: Improving llm-authored agent skills via trace-conditioned skill revision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.657137Z"},"links":{"cited_paper":"/paper/2606.01139","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:a67e45006144e9498c9d7d0436cb18e235a9e95319d8970e15f11353c4419c04","observation_id":"9521f4d7-2342-4664-8f4d-1a88f3503ef3","resolution":{"observed_at":"2026-08-01T04:28:46.657137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.18821","last_updated":"2026-05-19T07:37:46Z","snapshot_observed_at":"2026-08-02T13:01:02.442028Z","submitted_at":"2025-10-21T17:19:35Z","title":"Search Self-play: Pushing the Frontier of Agent Capability without Supervision","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.18821","snapshot_observed_at":"2026-08-01T04:28:46.810522Z","title":"Search self-play: Pushing the frontier of agent capability without supervision.arXiv preprint arXiv:2510.18821,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.810522Z"},"links":{"cited_paper":"/paper/2510.18821","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:1fe0ef652e4d44e40db53824e26b2ebf85fcd5d43f94cea6fbccc819dc746f34","observation_id":"e0bef9a0-5c8e-4d82-9306-3bcf9227c841","resolution":{"observed_at":"2026-08-01T04:28:46.810522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.02268","last_updated":"2026-05-15T09:15:24Z","snapshot_observed_at":"2026-07-06T22:51:44.663367Z","submitted_at":"2026-04-02T17:03:05Z","title":"SKILL0: In-Context Agentic Reinforcement Learning for Skill Internalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.02268","snapshot_observed_at":"2026-08-01T04:28:46.952475Z","title":"Skill0: In-context agentic reinforcement learning for skill internalization.arXiv preprint arXiv:2604.02268,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.952475Z"},"links":{"cited_paper":"/paper/2604.02268","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:d535c0e0af41f9478485ef07c4319f6717bdf5141aa31bf0897b59f20b64d1bf","observation_id":"d173e1cc-a592-4c26-a775-d1566ce09ab8","resolution":{"observed_at":"2026-08-01T04:28:46.952475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.01869","last_updated":"2026-05-28T12:51:20Z","snapshot_observed_at":"2026-08-03T05:35:29.225447Z","submitted_at":"2026-02-02T09:43:12Z","title":"Skill-Pro: Learning Reusable Skills from Experience via Non-Parametric PPO for LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.01869","snapshot_observed_at":"2026-08-01T04:28:47.064528Z","title":"Procmem: Learning reusable procedural memory from experience via non-parametric ppo for llm agents.arXiv preprint arXiv:2602.01869,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.064528Z"},"links":{"cited_paper":"/paper/2602.01869","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:006cd19811cb8569e0a88bcb8a97e8fc915fa621feb864d7e82a53d7ad267f25","observation_id":"4d74f686-e46e-4e04-80d0-371b74be435c","resolution":{"observed_at":"2026-08-01T04:28:47.064528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:47.217091Z","title":"Better alignment with instruction back-and-forth translation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.217091Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:9ee6a2252df923acf6d6fb94b88315b62d2a18cc807183ea26143df8a79073b7","observation_id":"91c76fc2-5099-44d7-848c-2a4df08b4823","resolution":{"observed_at":"2026-08-01T04:28:47.217091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.25158","last_updated":"2026-06-04T12:51:52Z","snapshot_observed_at":"2026-08-06T10:33:38.415457Z","submitted_at":"2026-03-26T08:26:38Z","title":"Trace2Skill: Distill Trajectory-Local Lessons into Transferable Agent Skills","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.25158","snapshot_observed_at":"2026-08-01T04:28:47.335859Z","title":"Trace2skill: Distill trajectory-local lessons into transferable agent skills.arXiv preprint arXiv:2603.25158,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.335859Z"},"links":{"cited_paper":"/paper/2603.25158","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:6d0b9c182d242950e226206b5cbbfd629f460f8c1a7a8b430a9f7bc3149504c0","observation_id":"c88c5bb2-7079-4a47-a693-2a4933a77eea","resolution":{"observed_at":"2026-08-01T04:28:47.335859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20309","last_updated":"2024-10-01T21:28:29Z","snapshot_observed_at":"2026-07-06T18:22:53.072592Z","submitted_at":"2024-05-30T17:52:36Z","title":"Large Language Models Can Self-Improve At Web Agent Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20309","snapshot_observed_at":"2026-08-01T04:28:47.436292Z","title":"Largelanguagemodelscanself-improveatwebagenttasks.arXivpreprintarXiv:2405.20309,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.436292Z"},"links":{"cited_paper":"/paper/2405.20309","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:151949cd9f8041aba63025638b65430494bfc1f0d7754362621de3357d0812de","observation_id":"15964e59-f02a-42c4-85d8-0bad85c80133","resolution":{"observed_at":"2026-08-01T04:28:47.436292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:47.552765Z","title":"Infobench: Evaluating instruction following ability in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.552765Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:d9b4e6f47f483c3858c7343820f8271a4ef2e2b75e0fb1b348e7de77ffe1ba93","observation_id":"9f32634f-2afc-489b-866f-bf837084d20a","resolution":{"observed_at":"2026-08-01T04:28:47.552765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:47.668921Z","title":"Sentence-bert: Sentence embeddings using siamese bert-networks","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.668921Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:a4b45de153f6f87ec8d386fea71e4f7433f32fab2212c9eaef5488c537c5410c","observation_id":"3f8aee4f-fa73-4196-8831-2197fb3ef672","resolution":{"observed_at":"2026-08-01T04:28:47.668921Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-01T04:28:47.878934Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.878934Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:244d444e23a670b08f93cbde6479ba6d120ada7f61d33b4e2b733578f65cc12a","observation_id":"8ef2ee6f-18ea-4836-bf27-e9b421aa3821","resolution":{"observed_at":"2026-08-01T04:28:47.878934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.03964","last_updated":"2026-04-05T05:02:18Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T05:02:18Z","title":"SKILLFOUNDRY: Building Self-Evolving Agent Skill Libraries from Heterogeneous Scientific Resources","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.03964","snapshot_observed_at":"2026-08-01T04:28:48.022530Z","title":"Skill- foundry: Building self-evolving agent skill libraries from heterogeneous scientific resources.arXiv preprint arXiv:2604.03964,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.022530Z"},"links":{"cited_paper":"/paper/2604.03964","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:ac856391ec3867f011a8fef4fbb856e92aad604a208265a40724c26e158d28cd","observation_id":"9d9e1a79-94e1-4133-bf5f-48e89bcb0001","resolution":{"observed_at":"2026-08-01T04:28:48.022530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14387","last_updated":"2024-06-03T17:47:30Z","snapshot_observed_at":"2026-07-06T18:03:51.687441Z","submitted_at":"2024-04-22T17:43:23Z","title":"A Survey on Self-Evolution of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14387","snapshot_observed_at":"2026-08-01T04:28:48.253415Z","title":"A survey on self-evolution of large language models.arXiv preprint arXiv:2404.14387,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.253415Z"},"links":{"cited_paper":"/paper/2404.14387","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:8b27c650ad83fb5ef721b5ad0c21fc3d276114a020047288303edb23f0710144","observation_id":"01050069-831c-4d3c-bc99-413d35f3e41b","resolution":{"observed_at":"2026-08-01T04:28:48.253415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:48.304531Z","title":"Code-a1: Adversarial evolving of code llm and test llm via reinforcement learning.arXiv preprint arXiv:2603.15611, 2026a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.304531Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:fa568f4a18b07d9384ccc4abd7a3ef88ff38ee4af7fec01794c244e6545b9f3c","observation_id":"89ede134-3d16-42ad-ada1-d4ac6ac62866","resolution":{"observed_at":"2026-08-01T04:28:48.304531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.18552","last_updated":"2026-06-02T06:06:43Z","snapshot_observed_at":"2026-08-03T15:02:08.953642Z","submitted_at":"2025-12-21T00:49:40Z","title":"Toward Training Superintelligent Software Agents through Self-Play SWE-RL","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.18552","snapshot_observed_at":"2026-08-01T04:28:48.430179Z","title":"Toward training superintelligent software agents through self-play swe-rl","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.430179Z"},"links":{"cited_paper":"/paper/2512.18552","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:11b06ab39da513dc95304f15eec25b4fd52e6e769d65095ffef1d5f181ee7496","observation_id":"36cfbba1-346e-4c22-a86c-f7d320e895ff","resolution":{"observed_at":"2026-08-01T04:28:48.430179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:48.564168Z","title":"Propose, solve, verify: Self-play through formal verification.arXiv preprint arXiv:2512.18160,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.564168Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:db0ac63260c050e73e2d8a3f4cf01a3a9c22470245e30dba73ff4a74e9d742cb","observation_id":"d0351b3c-df92-4b77-b934-d95962ece4e1","resolution":{"observed_at":"2026-08-01T04:28:48.564168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:48.732010Z","title":"Wizardlm: Empowering large pre-trained language models to follow complex instructions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.732010Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:dc8b0897668f32f0dceefa2cc1277a1abfdc888684eca273ba6e4c50ae738f06","observation_id":"6c223b5c-34cd-47a6-9283-f3f904b07b23","resolution":{"observed_at":"2026-08-01T04:28:48.732010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.06741","last_updated":"2026-06-04T21:55:48Z","snapshot_observed_at":"2026-07-06T23:46:28.942186Z","submitted_at":"2026-06-04T21:55:48Z","title":"OpenSkill: Open-World Self-Evolution for LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.06741","snapshot_observed_at":"2026-08-01T04:28:48.825254Z","title":"Genius: A generalizable and purely unsupervised self-training framework for advanced reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.825254Z"},"links":{"cited_paper":"/paper/2606.06741","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:36bb370c778caa31c094138d63ee4ddd4eb0cb0d4f09b1fc1677aa93be04a2ce","observation_id":"7d81a9a8-731e-4485-b7f3-514beb42fd7a","resolution":{"observed_at":"2026-08-01T04:28:48.825254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-01T04:28:48.911778Z","title":"Qwen3 technical report.arXiv preprint arXiv:2505.09388, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.911778Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:399316322125cc4aae2fa84ef1e88a028565754f42c1c05e17d214918be38eb6","observation_id":"1f323bd6-937c-496f-bcba-61d5d76ad2fc","resolution":{"observed_at":"2026-08-01T04:28:48.911778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.23751","last_updated":"2025-09-03T14:36:00Z","snapshot_observed_at":"2026-08-06T10:28:53.529265Z","submitted_at":"2025-07-31T17:38:50Z","title":"CoT-Self-Instruct: Building high-quality synthetic prompts for reasoning and non-reasoning tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.23751","snapshot_observed_at":"2026-08-01T04:28:48.960183Z","title":"Cot-self-instruct: Building high-quality synthetic prompts for reasoning and non-reasoning tasks.arXiv preprint arXiv:2507.23751, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.960183Z"},"links":{"cited_paper":"/paper/2507.23751","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:df84e95c1ffc518deb219947ea685fdb4f29570e46f67d899461f61905ebfa23","observation_id":"0e14d102-216a-41ad-a1c2-4695e8cd40b1","resolution":{"observed_at":"2026-08-01T04:28:48.960183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19590","last_updated":"2026-05-16T23:13:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-26T07:01:06Z","title":"Learning to Reason without External Rewards","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19590","snapshot_observed_at":"2026-08-01T04:28:49.009194Z","title":"Learning to reason without external rewards.arXiv preprint arXiv:2505.19590,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.009194Z"},"links":{"cited_paper":"/paper/2505.19590","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:4992708723cc060d65ae82be862d014ce9d9d8517e634e7fb447ec7c60e7f7f1","observation_id":"35acc60b-73d9-4046-8363-e7e252ca2fd6","resolution":{"observed_at":"2026-08-01T04:28:49.009194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.20087","last_updated":"2026-04-22T01:07:37Z","snapshot_observed_at":"2026-07-06T23:06:35.507008Z","submitted_at":"2026-04-22T01:07:37Z","title":"SkillLearnBench: Benchmarking Continual Learning Methods for Agent Skill Generation on Real-World Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.20087","snapshot_observed_at":"2026-08-01T04:28:49.067525Z","title":"Skilllearnbench: Benchmarking continual learning methods for agent skill generation on real-world tasks.arXiv preprint arXiv:2604.20087,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.067525Z"},"links":{"cited_paper":"/paper/2604.20087","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:e786755916bc80954a58e9e5a2ddd21bd7e10f0024c4db65665bc5dde7ef3601","observation_id":"f798bf16-0179-4e87-afc2-09e4b6e5797d","resolution":{"observed_at":"2026-08-01T04:28:49.067525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-01T04:28:49.133550Z","title":"Instruction-following evaluation for large language models.arXiv preprint arXiv:2311.07911,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.133550Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:2060c18e38e9f3e5f100e67728fe2c21e331442e6d9cfc4f14102bb0774bb1fe","observation_id":"6cf073f6-9c81-4cd5-8578-a9cf77f80205","resolution":{"observed_at":"2026-08-01T04:28:49.133550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:49.194675Z","title":"Webarena: A realistic web environment for building autonomous agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.194675Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:0e50080433cc8b00a6c9f9836d8c0e8b8897132d166ba85c6449c7cfe49d8e6c","observation_id":"bc7c0064-307a-4281-b5c6-0dc9d1a4f021","resolution":{"observed_at":"2026-08-01T04:28:49.194675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.18693","last_updated":"2026-05-18T17:28:36Z","snapshot_observed_at":"2026-08-06T06:41:28.384864Z","submitted_at":"2026-05-18T17:28:36Z","title":"SkillGenBench: Benchmarking Skill Generation Pipelines for LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.18693","snapshot_observed_at":"2026-08-01T04:28:49.274654Z","title":"Skillgenbench: Benchmarking skill generation pipelines for llm agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.274654Z"},"links":{"cited_paper":"/paper/2605.18693","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:330d0649bc500e5b9c431cd8c50485b159f15c74e305991f8fba6a32baa0f1a2","observation_id":"defb9909-59c5-48be-a39f-08b213d06789","resolution":{"observed_at":"2026-08-01T04:28:49.274654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:49.329638Z","title":"project updates","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.329638Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:09afe6c0f6add856660d5ae1e35b2ff5f335bd1151fc4cfdd4fee99743211fc5","observation_id":"36b44770-e84a-47f1-98f5-5dafee72db1f","resolution":{"observed_at":"2026-08-01T04:28:49.329638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:49.464315Z","title":"schedule a meeting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:49.464315Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:174c16a711bf757d9b8cca895481c4fb8098031d7a67d82e52e260bd02519607","observation_id":"f2175526-f46e-4dc3-90fa-b52d953befb4","resolution":{"observed_at":"2026-08-01T04:28:49.464315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.05808","last_updated":"2026-04-17T17:24:06Z","snapshot_observed_at":"2026-08-02T19:53:18.484273Z","submitted_at":"2026-01-09T14:32:06Z","title":"EnvScaler: Scaling Tool-Interactive Environments for LLM Agent via Programmatic Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.05808","snapshot_observed_at":"2026-08-01T04:28:48.135993Z","title":"Envscaler: Scaling tool-interactive environments for llm agent via programmatic synthesis.arXiv preprint arXiv:2601.05808,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:48.135993Z"},"links":{"cited_paper":"/paper/2601.05808","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:5015506801e9473befbb300301b5da9dc4d5796f651233c83334dbb084be7168","observation_id":"3d419ccf-b96f-4864-bf73-32ada2d1824f","resolution":{"observed_at":"2026-08-01T04:28:48.135993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.10084","last_updated":"2019-08-27T08:50:17Z","snapshot_observed_at":"2026-07-06T08:17:05.681370Z","submitted_at":"2019-08-27T08:50:17Z","title":"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.10084","snapshot_observed_at":"2026-08-01T04:28:47.805036Z","title":null,"venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:47.805036Z"},"links":{"cited_paper":"/paper/1908.10084","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:5356b6331c9dd0ded58481cb6abb62214399b115f9de185d80084264a577ff77","observation_id":"3b32beb6-35e5-4885-bc22-505a1fea3841","resolution":{"observed_at":"2026-08-01T04:28:47.805036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:46.086644Z","title":"Language self-play for data-free training.arXiv preprint arXiv:2509.07414,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:46.086644Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:fdf535fc4fb958d199051387e8d10cd4dbec5fd41374ddfedb134a1e3a76692b","observation_id":"27ae9a1c-8fde-48eb-a40e-cb7097003ffc","resolution":{"observed_at":"2026-08-01T04:28:46.086644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.11605","last_updated":"2025-03-16T09:43:15Z","snapshot_observed_at":"2026-08-05T08:37:25.351024Z","submitted_at":"2024-12-16T09:47:43Z","title":"SPaR: Self-Play with Tree-Search Refinement to Improve Instruction-Following in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.11605","snapshot_observed_at":"2026-08-01T04:28:45.255280Z","title":"Spar: Self-play with tree-search refinement to improve instruction-following in large language models.arXiv preprint arXiv:2412.11605, 2024a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.255280Z"},"links":{"cited_paper":"/paper/2412.11605","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:d9cceeb971827af698e8f45a521ce53e34d72691c15e21b5feb93dc86d8c28cd","observation_id":"064ca75f-2f59-4c0e-88c8-753332ffd3d5","resolution":{"observed_at":"2026-08-01T04:28:45.255280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.03980","last_updated":"2026-06-02T17:56:57Z","snapshot_observed_at":"2026-07-06T23:44:09.692359Z","submitted_at":"2026-06-02T17:56:57Z","title":"Skill-RM: Unifying Heterogeneous Evaluation Criteria via Agent Skill","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.03980","snapshot_observed_at":"2026-08-01T04:28:45.019999Z","title":"Skill-rm: Unifying heterogeneous evaluation criteria via agent skill.arXiv preprint arXiv:2606.03980, 2026a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.019999Z"},"links":{"cited_paper":"/paper/2606.03980","citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:2e7d2f5c13094e570e02bdf55bc9713a75fa1fd9263858742667261a85859364","observation_id":"dd69018d-9cdf-4f1b-b65d-2d2cc94df016","resolution":{"observed_at":"2026-08-01T04:28:45.019999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T04:28:45.367391Z","title":"Self- play with execution feedback: Improving instruction-following capabilities of large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-01T04:28:45.367391Z"},"links":{"citing_paper":"/paper/2607.22529"},"observation_digest":"sha256:f20a6cd732856fd79ffca16f8bcfeadee403d50c5e2693af284fb6380ec590d6","observation_id":"232f5c1d-59d6-4eaa-a412-8853239a5168","resolution":{"observed_at":"2026-08-01T04:28:45.367391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.22529","last_updated":"2026-07-24T17:59:22Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-01T04:28:42.700487Z","submitted_at":"2026-07-24T17:59:22Z","title":"Skill Self-Play: Pushing the Frontier of LLM Capability with Co-Evolving Skills"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 0 inbound Pith citation observations for arXiv:2607.22529."}