{"as_of":"2026-08-16T16:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a07f59bdab4d529810c3ed98ffd0bcdedd7a1db8e082b8912e1ddde4b3cf5691","coverage":[{"denominator":75,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":75,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T00:15:04.780467Z","state":"measured"},{"denominator":75,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":75,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.12253/citation-record","integrity":"/paper/2608.12253/integrity","json":"/paper/2608.12253/citation-record.json","paper":"/paper/2608.12253"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-16T00:15:04.453312Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.453312Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:155cd703fbda48317bdaeeae58f3614a6b77e0f79787585bc6c6fa124b14d934","observation_id":"ea338534-eff5-4f60-b471-4449fd62e289","resolution":{"observed_at":"2026-08-16T00:15:04.453312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-08-13T12:34:54.476684Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-16T00:15:04.460262Z","title":"Understanding r1-zero-like training: A critical perspective, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.460262Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:e2f2025ac09a7650d217c09b1f549ec4c357a9b3b2520651c478bcfe6461fd83","observation_id":"8c737c19-cd18-448f-90d1-888e0e7f4e82","resolution":{"observed_at":"2026-08-16T00:15:04.460262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-16T00:15:04.465639Z","title":"Dapo: An open-source llm reinforcement learning system at scale, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.465639Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:e65ea86648af1474ab1fd62cf214bfad316fb87f01a396de2917678a648b6ea2","observation_id":"a79b4430-5332-4d40-8254-c46cd54e5875","resolution":{"observed_at":"2026-08-16T00:15:04.465639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11536","last_updated":"2025-04-17T16:46:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-15T18:10:22Z","title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.11536","snapshot_observed_at":"2026-08-16T00:15:04.471444Z","title":"Retool: Reinforcement learning for strategic tool use in llms, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.471444Z"},"links":{"cited_paper":"/paper/2504.11536","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:ef0d8b0daffc50ac8b46ec8c6a5d073865d1005a7569392aed63910246f63302","observation_id":"9878a06f-50fd-4c68-a7f5-23ce74c7ed8b","resolution":{"observed_at":"2026-08-16T00:15:04.471444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09516","last_updated":"2025-08-05T19:08:38Z","snapshot_observed_at":"2026-08-15T13:17:00.526689Z","submitted_at":"2025-03-12T16:26:39Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09516","snapshot_observed_at":"2026-08-16T00:15:04.476116Z","title":"Search-r1: Training llms to reason and leverage search engines with reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.476116Z"},"links":{"cited_paper":"/paper/2503.09516","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:6e733ca207aceb9d6417680ac6d1acf4ae82e49a55c7258d9bbfcbe27e8f4912","observation_id":"bc2336a1-ff6b-40d7-9e78-707c7ae155b0","resolution":{"observed_at":"2026-08-16T00:15:04.476116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.21798","last_updated":"2025-05-21T17:21:45Z","snapshot_observed_at":"2026-08-11T02:21:44.551484Z","submitted_at":"2025-04-30T16:56:06Z","title":"SWE-smith: Scaling Data for Software Engineering Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.21798","snapshot_observed_at":"2026-08-16T00:15:04.480953Z","title":"Jimenez, Alexander Wettig, Kabir Khandpur, Yanzhe Zhang, Binyuan Hui, Ofir Press, Ludwig Schmidt, and Diyi Yang","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.480953Z"},"links":{"cited_paper":"/paper/2504.21798","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:ef24ca8c58bfd635abc3167ac41d7523a8b6bb10fc3b89b4fb8b78cba109017e","observation_id":"66688e70-7873-4b78-b699-10b0eabd0654","resolution":{"observed_at":"2026-08-16T00:15:04.480953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18449","last_updated":"2025-12-01T00:16:59Z","snapshot_observed_at":"2026-08-13T07:37:18.494967Z","submitted_at":"2025-02-25T18:45:04Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18449","snapshot_observed_at":"2026-08-16T00:15:04.486458Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.486458Z"},"links":{"cited_paper":"/paper/2502.18449","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:12eaa93286fb582129af5ecb2e62cd37cdc509011fac73542a4ce3fd7174de41","observation_id":"757ed77b-732e-4bc0-baab-0d2158e02df5","resolution":{"observed_at":"2026-08-16T00:15:04.486458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.996542Z","title":"UserRL: Training Interactive User-Centric Agent via Reinforcement Learning, September","venue":null,"work_id":"d11d25ce-959e-4174-8c5b-48de2ae39680","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.491426Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:cdf4cd3e292a2812b4c9d82143d209e192b4319bd1ee5cc192300eae6e104029","observation_id":"a71afce6-d652-4621-aa52-428c2a44339f","resolution":{"observed_at":"2026-08-16T00:15:07.001260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.500306Z","title":"TOM-SWE: User Mental Modeling For Software Engineering Agents, October 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.500306Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:23a49c4877ba5dded54a8f3dda278b04eeae8cce8723b3b00fc922739949c445","observation_id":"8fc7d44e-c8c1-40e2-8039-afce0e969ef4","resolution":{"observed_at":"2026-08-16T00:15:04.500306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.504225Z","title":"HumanLM: Simulating Users with State Alignment Beats Response Imitation, February 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.504225Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:be82dd2223b4b09e73d7bf64ed59960eef7e66c9bc2835d7581a7421841622f9","observation_id":"56c06e39-cbb7-4a26-b23a-53b036d7c4f2","resolution":{"observed_at":"2026-08-16T00:15:04.504225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-08-14T06:34:01.114459Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-08-16T00:15:04.507966Z","title":"τ 2-Bench: Evaluating Conversational Agents in a Dual-Control Environment, June 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.507966Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:34a1541e2ab907d5d6db1f78a415547d225ea4528cf14f631745d94d10c97f55","observation_id":"558deb1f-850c-47cf-a278-ee375086da82","resolution":{"observed_at":"2026-08-16T00:15:04.507966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.511940Z","title":"Lieberwirth, Xinkai Yu, Yicheng Fu, Michael J","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.511940Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:dd84c1afe553524e402cee408f3c12147370bde5c7e77025d814ada6969f338c","observation_id":"3e3bdbe8-e2ec-455f-8129-2d9b8a6e7f86","resolution":{"observed_at":"2026-08-16T00:15:04.511940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.982372Z","title":"Position: Humans are missing from ai coding agent research","venue":null,"work_id":"db8d16fc-f2f8-4e38-92f7-03e0d2aaf8dd","year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.516143Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:af3beb7c986725bd040ca156326f44c321425bd1769c4fb58ebc15c8f1699452","observation_id":"a3ec8aee-1627-4f78-a2f2-0d710c2bc7d2","resolution":{"observed_at":"2026-08-16T00:15:06.986895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.967907Z","title":"Persuasion for good: Towards a personalized persuasive dialogue system for social good","venue":null,"work_id":"b546bbe9-ed44-429d-9699-aeac457e9c4d","year":2019},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.520335Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:6752f521d8dae20fd17c8f6b22cd31525a941bccccf0d62067b9a4c8bf89ef88","observation_id":"83b9cb85-b2d3-435b-86c3-92d328846d39","resolution":{"observed_at":"2026-08-16T00:15:06.972671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.524125Z","title":"Consistently Simulating Human Personas with Multi-Turn Reinforcement Learning, October","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.524125Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:0da59690cf8f762017737e28182dbcecb7b3556422be91b2faf09432f76b3a5b","observation_id":"73bb10d5-7452-4c18-91d9-21ef361ab848","resolution":{"observed_at":"2026-08-16T00:15:04.524125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.533329Z","title":"Bernstein","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.533329Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:55f414c12f3f10c3542a83a126070f44e77321f75209efa6f67fcd53123e825c","observation_id":"354ddfcc-ccf8-446b-a58b-ad800dfc1ed8","resolution":{"observed_at":"2026-08-16T00:15:04.533329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.528529Z","title":"arXiv:2511.00222 [cs]","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.528529Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:41f01cd48073d1e401dda0a5df1f34e179a5c20192dc79c389f36160865a7de2","observation_id":"81d1c642-83c0-4b21-a540-5ea5a779176d","resolution":{"observed_at":"2026-08-16T00:15:04.528529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.542316Z","title":"Sotopia-RL: Reward Design for Social Intelligence, October 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.542316Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:61d1d70ca04dac17261b70fd975a73a3f41016c8a22bbbde3da96ab81280d482","observation_id":"1f1cec51-b85e-4999-b8a3-a21dde612dc8","resolution":{"observed_at":"2026-08-16T00:15:04.542316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10109","last_updated":"2026-06-28T22:37:51Z","snapshot_observed_at":"2026-08-13T12:33:17.230204Z","submitted_at":"2024-11-15T11:14:34Z","title":"LLM Agents Grounded in Self-Reports Enable General-Purpose Simulation of Individuals","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10109","snapshot_observed_at":"2026-08-16T00:15:04.537288Z","title":"Zou, Jonne Kamphorst, Niles Egan, Aaron Shaw, Benjamin Mako Hill, Carrie Cai, Meredith Ringel Morris, Percy Liang, Robb Willer, and Michael S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.537288Z"},"links":{"cited_paper":"/paper/2411.10109","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:e47efe4b448ea089d76ad038536fb08ccfe068599ffc377251900da54ad8411f","observation_id":"53ce8f94-5f2f-409d-80d2-36873209c9da","resolution":{"observed_at":"2026-08-16T00:15:04.537288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.551128Z","title":"Artificial Hivemind: The Open-Ended Homogeneity of Language Models (and Beyond), October 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.551128Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:fa40745bb2d4f2b506386a6bd0498ce7c5cd373bf9bb3c4fa2269cec7e04dcb9","observation_id":"4ece0c50-a8cc-47e5-b206-4a5bb9d7f39a","resolution":{"observed_at":"2026-08-16T00:15:04.551128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18232","last_updated":"2023-11-30T03:59:31Z","snapshot_observed_at":"2026-08-16T14:38:59.314544Z","submitted_at":"2023-11-30T03:59:31Z","title":"LMRL Gym: Benchmarks for Multi-Turn Reinforcement Learning with Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18232","snapshot_observed_at":"2026-08-16T00:15:04.546670Z","title":"Lmrl gym: Benchmarks for multi-turn reinforcement learning with language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.546670Z"},"links":{"cited_paper":"/paper/2311.18232","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:fd8f8e4ce007b42208babcc4b75ca37a6e537aece0a59ade89e2999c6daec2bd","observation_id":"0826ad6c-26d8-47a0-9f7d-84241ab7d5f9","resolution":{"observed_at":"2026-08-16T00:15:04.546670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.01171","last_updated":"2026-07-15T22:07:56Z","snapshot_observed_at":"2026-08-08T10:41:29.959685Z","submitted_at":"2025-10-01T17:55:37Z","title":"Verbalized Sampling: How to Mitigate Mode Collapse and Unlock LLM Diversity","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.01171","snapshot_observed_at":"2026-08-16T00:15:04.558740Z","title":"Tomz, Christopher D","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.558740Z"},"links":{"cited_paper":"/paper/2510.01171","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:068ea982c30c8f631546a91b61ab1e923973d3c28a22c81d88d97c0016f6c986","observation_id":"4c759508-dca8-4e29-9d72-620b09e13353","resolution":{"observed_at":"2026-08-16T00:15:04.558740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.555118Z","title":"KL-regularized reinforcement learning is designed to mode collapse, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.555118Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:93147a5f2532f5db4de15f2a7f7f1d2881b99600a9e1695234bba71420437725","observation_id":"a82446fb-cdb2-44c9-94d3-738c206b6ebc","resolution":{"observed_at":"2026-08-16T00:15:04.555118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.568741Z","title":"Chasing moving targets with online self-play reinforcement learning for safer language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.568741Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:ae727625936901c7a2b0c2e0d85ccb4e28ec0b82680aa5d8c79cc4ebe237aaff","observation_id":"5187323f-70e8-4871-b27b-b894c734c55f","resolution":{"observed_at":"2026-08-16T00:15:04.568741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.564081Z","title":"Natural emergent misalignment from reward hacking in production RL, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.564081Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:2952f344522412898f16714823dfc973a1ee9c94f861ecd3d9a02a3d15af369d","observation_id":"1b5f4f66-123a-4944-8531-c3cb8d585c11","resolution":{"observed_at":"2026-08-16T00:15:04.564081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.577287Z","title":"Williams","venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.577287Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:3f2844f933050ab440d19f43e2f9f9851e0ea6f35bacf55645863423fa3511ff","observation_id":"8a2e3376-0928-46c9-9f6e-382e2fc0b6d0","resolution":{"observed_at":"2026-08-16T00:15:04.577287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.573268Z","title":"SPIRAL: Self-Play on Zero-Sum Games Incentivizes Reasoning via Multi-Agent Multi-Turn Reinforcement Learning, July 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.573268Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:1112f2f3bff7b17229784931b4f1f739a43e886a7026c577f70e861dd17a5c20","observation_id":"78310470-81af-40b0-8f96-2253d485a236","resolution":{"observed_at":"2026-08-16T00:15:04.573268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.590001Z","title":"Flipping the Dialogue: Training and Evaluating User Language Models, October 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.590001Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:06ade843297e1e6c74aaf81a44380b06df6d8506ef031be882b696b38b1b4cc4","observation_id":"79b4edda-9754-47ec-bf58-3e997bd46958","resolution":{"observed_at":"2026-08-16T00:15:04.590001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.937645Z","title":"Noveltybench: Evaluating language models for humanlike diversity,","venue":null,"work_id":"3a0def0a-1f8c-4c4a-8a9b-215fd3c87046","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.581498Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:35297e51124c6025edf1ec5d8cbf238ac63217fcab1c3e5c03265555044fea50","observation_id":"66956b33-e5c5-4f58-b112-8a0619171687","resolution":{"observed_at":"2026-08-16T00:15:06.942798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05228","last_updated":"2025-08-09T18:33:13Z","snapshot_observed_at":"2026-08-16T12:43:18.364897Z","submitted_at":"2025-04-07T16:14:23Z","title":"NoveltyBench: Evaluating Language Models for Humanlike Diversity","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.05228","snapshot_observed_at":"2026-08-16T00:15:04.585634Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.585634Z"},"links":{"cited_paper":"/paper/2504.05228","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:aeb034707028e3519181a15b6ee36ab5da6647a2a79a9d207ad2063d8c0fcfa6","observation_id":"c524adc3-b61f-4a28-ad18-499a4ec70cbd","resolution":{"observed_at":"2026-08-16T00:15:04.585634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.602450Z","title":"SPICE: Self-Play In Corpus Environments Improves Reasoning, October 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.602450Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:edebf6105ed569e9f4d060f73459a3a209fdfed4035f530c78f2586399d0b9fe","observation_id":"500f716f-e128-4c32-a457-ba34fed773ea","resolution":{"observed_at":"2026-08-16T00:15:04.602450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.11245","last_updated":"2026-07-31T22:23:28Z","snapshot_observed_at":"2026-08-15T10:27:16.529158Z","submitted_at":"2026-03-11T19:12:31Z","title":"Mind the Sim2Real Gap in User Simulation for Agentic Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.11245","snapshot_observed_at":"2026-08-16T00:15:04.593555Z","title":"Mind the sim2real gap in user simulation for agentic tasks, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.593555Z"},"links":{"cited_paper":"/paper/2603.11245","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:5bd1978ca010a8f5f60fad9cb1bfb66056c5e136b76e7d0cef62a800f69bd4ec","observation_id":"f747676c-b894-444b-9219-ced9c253466b","resolution":{"observed_at":"2026-08-16T00:15:04.593555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.07847","last_updated":"2026-05-08T15:09:25Z","snapshot_observed_at":"2026-08-15T00:14:04.828373Z","submitted_at":"2026-05-08T15:09:25Z","title":"Measuring and Mitigating the Distributional Gap Between Real and Simulated User Behaviors","version":1},"cited_work":{"arxiv_id":"2605.07847","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.07847","snapshot_observed_at":"2026-08-16T00:15:05.777517Z","title":"Measuring and Mitigating the Distributional Gap Between Real and Simulated User Behaviors","venue":"cs.CL","work_id":"bd4508de-0bc6-4666-bc7e-154d8524b313","year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.598211Z"},"links":{"cited_paper":"/paper/2605.07847","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:3d547196318e832419aef28d61c58ee03f9e5380633d0dd69856dcaed468b536","observation_id":"29fce8ca-4b7b-4ce2-b59d-c7f8cc5b66d9","resolution":{"observed_at":"2026-08-16T00:15:05.782484Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.06268","last_updated":"2026-04-07T04:29:41Z","snapshot_observed_at":"2026-08-13T19:20:01.823209Z","submitted_at":"2026-04-07T04:29:41Z","title":"RAGEN-2: Reasoning Collapse in Agentic RL","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.06268","snapshot_observed_at":"2026-08-16T00:15:04.613847Z","title":"Ragen-2: Reasoning collapse in agentic rl, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.613847Z"},"links":{"cited_paper":"/paper/2604.06268","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:9e506280bc0bce0ecba7a650f1a7f46b5c72016ef06fa8f5d9463935763d7965","observation_id":"e5a69584-fcaf-4d86-9026-f1b8ba864561","resolution":{"observed_at":"2026-08-16T00:15:04.613847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.923757Z","title":"Hi- erarchical agenda reasoning for strategic multi-turn dialogue agents","venue":null,"work_id":"bdbc0b3e-7a77-4e89-a35d-779880ca76a3","year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.606356Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:73afdd09eaefc0f1f464a42d2725563b621a81ec1de5dc80ed62e414ce0a5a8a","observation_id":"35736ef2-fde6-4f85-8a06-f8422e014cf9","resolution":{"observed_at":"2026-08-16T00:15:06.927989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.610256Z","title":"Llm probability concentration: How alignment shrinks the generative horizon, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.610256Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:caf9e1daa50bf0112a16a2bf4fd9faf358e462bee223a8990515fcb32701ac64","observation_id":"7a78fb76-ff17-4df9-9991-59e7de1a0857","resolution":{"observed_at":"2026-08-16T00:15:04.610256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18669","last_updated":"2025-08-26T04:26:29Z","snapshot_observed_at":"2026-08-13T12:45:20.389437Z","submitted_at":"2025-08-26T04:26:29Z","title":"MUA-RL: Multi-turn User-interacting Agent Reinforcement Learning for agentic tool use","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.18669","snapshot_observed_at":"2026-08-16T00:15:04.627801Z","title":"MUA-RL: Multi-turn User-interacting Agent Reinforcement Learning for agentic tool use, August 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.627801Z"},"links":{"cited_paper":"/paper/2508.18669","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:17355b28a25e3f752abbeedfa82e25c76a0eee592440097de64b481bc781fe8a","observation_id":"d46331dc-b1ae-4f63-97d0-e5f686348cc2","resolution":{"observed_at":"2026-08-16T00:15:04.627801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24698","last_updated":"2026-04-27T17:01:48Z","snapshot_observed_at":"2026-08-13T20:49:52.724746Z","submitted_at":"2026-04-27T17:01:48Z","title":"The Chameleon's Limit: Investigating Persona Collapse and Homogenization in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.24698","snapshot_observed_at":"2026-08-16T00:15:04.617997Z","title":"Zhang, Chenghao Yang, Ningshan Ma, Weihao Xuan, and Jen tse Huang","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.617997Z"},"links":{"cited_paper":"/paper/2604.24698","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:d9bca827db440ed76873875a042c47c05d0ceed418e4f941d8e9f6a3203f8cf6","observation_id":"ad985377-0b6b-436b-8bb0-52cfc35b826b","resolution":{"observed_at":"2026-08-16T00:15:04.617997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02234","last_updated":"2025-06-05T15:12:55Z","snapshot_observed_at":"2026-08-16T12:44:24.688160Z","submitted_at":"2025-04-03T03:01:26Z","title":"LLM Social Simulations Are a Promising Research Method","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02234","snapshot_observed_at":"2026-08-16T00:15:04.623422Z","title":"Richardson, Austin C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.623422Z"},"links":{"cited_paper":"/paper/2504.02234","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:580bf795c7cb4a0a29dc3cb437bfb7ec2ddc8be86d229f2de951fecdb399f34b","observation_id":"f9dd7c70-72cd-4694-9e11-3fa92c1170f7","resolution":{"observed_at":"2026-08-16T00:15:04.623422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12894","last_updated":"2026-05-13T02:16:51Z","snapshot_observed_at":"2026-08-16T00:01:50.877069Z","submitted_at":"2026-05-13T02:16:51Z","title":"Beyond Cooperative Simulators: Generating Realistic User Personas for Robust Evaluation of LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.12894","snapshot_observed_at":"2026-08-16T00:15:04.642064Z","title":"Beyond cooperative simulators: Generating realistic user personas for robust evaluation of llm agents, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.642064Z"},"links":{"cited_paper":"/paper/2605.12894","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:8d079af6c5bd233c4fea4eaa8e516c3a5eedb07f9025218488d51a51069aff5b","observation_id":"fe626c99-dc7f-4f61-a654-bc4936757de8","resolution":{"observed_at":"2026-08-16T00:15:04.642064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.632848Z","title":"Training Proactive and Personalized LLM Agents, November 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.632848Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:fb517133a1abf2bd30ad0a41fe5ba2c99347602b20f677962d71c264b2e9d630","observation_id":"6e90aac5-60d9-4210-830d-76adf774ed7e","resolution":{"observed_at":"2026-08-16T00:15:04.632848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.637451Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.637451Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:60b350d563dd3438f8bf2409d4620d490422f114ad97c8b396505464a5bdc1f1","observation_id":"2a0c0564-38f9-4f7b-be63-d0b313474b43","resolution":{"observed_at":"2026-08-16T00:15:04.637451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.654081Z","title":"Enhancing personalized multi-turn dialogue with curiosity reward, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.654081Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:21a1a101fa156c6cde2eb7157ad519ac24cf37a5d1fe19f8f84bfe9ddc9fbec0","observation_id":"4085fdb1-4365-4202-aa57-da42c1199097","resolution":{"observed_at":"2026-08-16T00:15:04.654081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.09808","last_updated":"2026-05-10T23:06:24Z","snapshot_observed_at":"2026-08-16T10:00:09.673723Z","submitted_at":"2026-05-10T23:06:24Z","title":"Quantifying the Utility of User Simulators for Building Collaborative LLM Assistants","version":1},"cited_work":{"arxiv_id":"2605.09808","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.09808","snapshot_observed_at":"2026-08-16T00:15:05.234497Z","title":"Quantifying the Utility of User Simulators for Building Collaborative LLM Assistants","venue":"cs.CL","work_id":"4f683c85-b7ab-419f-b329-53db566255af","year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.646337Z"},"links":{"cited_paper":"/paper/2605.09808","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:9ec8c84a2cb350879338665c91117898ac1baa555a7b6ceee22c901e2c8a12cb","observation_id":"3ee0ed04-67ad-4600-82c0-53b2f33555f1","resolution":{"observed_at":"2026-08-16T00:15:05.239896Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.650372Z","title":"Non- collaborative user simulators for tool agents, September 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.650372Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:8bd88be5388f1465e24a96acce061b44ffc12199af140168f9b9aa7cad56c7e5","observation_id":"a8357c08-443e-43b1-b8ac-30337175e0d9","resolution":{"observed_at":"2026-08-16T00:15:04.650372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.06680","last_updated":"2019-12-13T19:56:40Z","snapshot_observed_at":"2026-08-13T14:13:07.807518Z","submitted_at":"2019-12-13T19:56:40Z","title":"Dota 2 with Large Scale Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.06680","snapshot_observed_at":"2026-08-16T00:15:04.669651Z","title":null,"venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.669651Z"},"links":{"cited_paper":"/paper/1912.06680","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:5ffafd8d7b3cc5815f97757d91f53a9e7b3c86d56746e2b741431faf00b45fae","observation_id":"ace82092-9a6c-4c0d-944a-246df5695d30","resolution":{"observed_at":"2026-08-16T00:15:04.669651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.08567","last_updated":"2026-04-27T22:45:19Z","snapshot_observed_at":"2026-08-13T00:03:12.081078Z","submitted_at":"2026-03-19T19:31:53Z","title":"Multi-User Large Language Model Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.08567","snapshot_observed_at":"2026-08-16T00:15:04.658345Z","title":"Bakker, and Jiaxin Pei","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.658345Z"},"links":{"cited_paper":"/paper/2604.08567","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:54ff2b02f25141437ca9afcd399b2a8e408ef92bec13404e38670afacef39122","observation_id":"070cc558-62ca-4e34-81d2-7ec53b558aab","resolution":{"observed_at":"2026-08-16T00:15:04.658345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.663523Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.663523Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:4a447b22546837690492e6105796fbe32ca8ddf9c4028fb1479448b0c962ad38","observation_id":"f82917b2-2ab5-4a64-b603-54e0e999b232","resolution":{"observed_at":"2026-08-16T00:15:04.663523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.909305Z","title":"Coevolving with the other you: Fine-tuning LLM with sequential cooperative multi-agent reinforcement learning","venue":null,"work_id":"718e259b-706b-4005-8908-3a8ca86895bc","year":2024},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.682545Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:73bb304b71f955c828f469ab739d3d6cc7326c1128c622c9577c432c3e171560","observation_id":"93466362-ff19-47fc-bd89-79c4fcf393b6","resolution":{"observed_at":"2026-08-16T00:15:06.914295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03335","last_updated":"2025-10-16T08:23:36Z","snapshot_observed_at":"2026-08-11T20:12:00.210052Z","submitted_at":"2025-05-06T09:08:00Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.03335","snapshot_observed_at":"2026-08-16T00:15:04.674667Z","title":"Absolute Zero: Reinforced Self- play Reasoning with Zero Data, October 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.674667Z"},"links":{"cited_paper":"/paper/2505.03335","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:eeaad1836f9ec222662ba867795d543398763de0e08bf71216d0f5b3e53d82f6","observation_id":"47ad6414-e8a4-4302-8dd4-6dac1276fd00","resolution":{"observed_at":"2026-08-16T00:15:04.674667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18872","last_updated":"2024-12-09T02:45:45Z","snapshot_observed_at":"2026-08-16T13:39:13.314550Z","submitted_at":"2024-06-27T03:52:35Z","title":"Efficacy of Language Model Self-Play in Non-Zero-Sum Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.18872","snapshot_observed_at":"2026-08-16T00:15:04.678569Z","title":"Efficacy of language model self-play in non- zero-sum games, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.678569Z"},"links":{"cited_paper":"/paper/2406.18872","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:9b17676eaaf8a2272650eeef20c51f2d52f96d4e753de5440cfa54cc10347d2f","observation_id":"0e0d651d-7a12-429a-b7ed-7ec8b4a1abfa","resolution":{"observed_at":"2026-08-16T00:15:04.678569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15478","last_updated":"2025-03-19T17:55:08Z","snapshot_observed_at":"2026-08-16T12:48:30.366139Z","submitted_at":"2025-03-19T17:55:08Z","title":"SWEET-RL: Training Multi-Turn LLM Agents on Collaborative Reasoning Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.15478","snapshot_observed_at":"2026-08-16T00:15:04.693596Z","title":"SWEET-RL: Training Multi-Turn LLM Agents on Collaborative Reasoning Tasks, March 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.693596Z"},"links":{"cited_paper":"/paper/2503.15478","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:f56f6de8a5f91c64710160173914e179676e597c54e10b57319d6fda2c809ca0","observation_id":"2c457f6f-b581-4af4-ac77-d68b18bbca7f","resolution":{"observed_at":"2026-08-16T00:15:04.693596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.896420Z","title":null,"venue":null,"work_id":"a07553ae-768f-4207-ac52-bdbd8f9705ad","year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.686224Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:d810aba5942290aeac6cc882927fdb351317fced9bc2425093ffb2997b2bb779","observation_id":"63c1f234-8256-4890-af30-af80c11a030d","resolution":{"observed_at":"2026-08-16T00:15:06.900484Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.883634Z","title":"Tool-R0: Self-evolving LLM agents for tool-learning from zero data, 2026","venue":null,"work_id":"4756802f-429b-4267-9e31-dab50b8c4089","year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.689937Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:2853a5586f5b6729f14cae41d1c7094dad514d5bbc690c749839fdcabfaebedf","observation_id":"d64f14a8-bfb1-4203-b8e9-ada6c586c6b1","resolution":{"observed_at":"2026-08-16T00:15:06.887983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.868415Z","title":"Manning, Stefano Ermon, and Chelsea Finn","venue":null,"work_id":"b5036398-f1dc-47ae-b797-025816dbbe98","year":2023},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.705959Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:344acc247d6fcd2d0daa5d692924ae14b89e631f733aa0fa1b325e3bf1593a00","observation_id":"d516b38c-3679-40e4-9459-4b727b4c1c38","resolution":{"observed_at":"2026-08-16T00:15:06.873951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.697681Z","title":"Natural Language Actor- Critic: Scalable Off-Policy Learning in Language Space, December 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.697681Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:aaa0f20ece9af780c1c3e9987391c83639fdd41dacd49e5e4c388a291131f836","observation_id":"a7a17a8e-a712-41c7-9b63-b7075dee5a45","resolution":{"observed_at":"2026-08-16T00:15:04.697681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.701587Z","title":"Ma, Seun Eisape, Ellie French, Tingting Du, Tianjiao Zhang, Alexander Koller, and Alane Suhr","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.701587Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:c62f262d9f4ff7b9061bef296a4849769e1a7077258def14e1bc9bf644242331","observation_id":"c1cb2f5c-9429-4353-a794-7ef920329924","resolution":{"observed_at":"2026-08-16T00:15:04.701587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-08-13T12:26:05.192883Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-08-16T00:15:04.719186Z","title":"Openai gym.arXiv preprint arXiv:1606.01540, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.719186Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:84de6b46707d80a2e8529a9bf5740cb2afb3da8cc528ded0a1c9009ada247d68","observation_id":"a0b5a150-ffbf-49ad-a1e3-8cbaf1a3e4aa","resolution":{"observed_at":"2026-08-16T00:15:04.719186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-16T00:15:04.710147Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.710147Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:a3c491dbb6efe9e3932b87e465ce3426c6ea9665faa6506005134d3e82fd625a","observation_id":"904f6f92-78bf-45fc-aa6b-145d9552b2d0","resolution":{"observed_at":"2026-08-16T00:15:04.710147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-08-15T20:26:32.102285Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-16T00:15:04.714785Z","title":"Proximal policy optimization algorithms.arXiv preprint arXiv:1707.06347, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.714785Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:6e65ebbd70006d1319079dac70dc97904bf63f8724f73751564e7f810863e8a5","observation_id":"3adb7e51-8235-488b-b820-79e5e5a13341","resolution":{"observed_at":"2026-08-16T00:15:04.714785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.731708Z","title":"slime: An llm post-training framework for rl scaling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.731708Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:2bbf3692ce3cd3c9baa918edb80fb272896d33a5be1d1a8264b51b1e49c48984","observation_id":"b3fcff93-69b1-487c-ba01-85caeb68a3b1","resolution":{"observed_at":"2026-08-16T00:15:04.731708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-08-11T10:42:54.633244Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07104","snapshot_observed_at":"2026-08-16T00:15:04.723393Z","title":"Gonzalez, Clark Barrett, and Ying Sheng","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.723393Z"},"links":{"cited_paper":"/paper/2312.07104","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:9e481c24677019899bef7f62e5b890cf8414e342204e4e8575426ff814424344","observation_id":"e1252219-a1fd-4789-85e7-d96bcc40214c","resolution":{"observed_at":"2026-08-16T00:15:04.723393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-08-12T10:50:46.357243Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-16T00:15:04.727651Z","title":"Megatron-lm: Training multi-billion parameter language models using model parallelism.CoRR, abs/1909.08053, 2019","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.727651Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:bab734bfffb21003b339e8a8c9eb137bc511cc5f69ea3add8e21f6f69589c38a","observation_id":"83045f16-1c71-4538-b8b6-a9be9f71207b","resolution":{"observed_at":"2026-08-16T00:15:04.727651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.743354Z","title":"Qwen3.5: Towards native multimodal agents, February 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.743354Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:e3ebdc386816fe7c3a39aec596857a80f17e3a1dfb753eabea6c00d578704a12","observation_id":"bbbdf98a-1e51-4c62-8d99-6498a90f2f81","resolution":{"observed_at":"2026-08-16T00:15:04.743354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.849399Z","title":"HybridFlow: A flexible and efficient RLHF framework","venue":null,"work_id":"0a788370-4602-46fc-bc75-5e29e1166ea5","year":2024},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.735504Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:1876069c47687e0a0fbebe0707be3cf10cf0bbd425fc90d668ad8e7314303fdf","observation_id":"7c53bbcc-df9b-4991-aaf0-e03e05a32f86","resolution":{"observed_at":"2026-08-16T00:15:06.853370Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.15565","last_updated":"2026-05-15T03:13:35Z","snapshot_observed_at":"2026-08-03T02:43:16.115285Z","submitted_at":"2026-05-15T03:13:35Z","title":"AstraFlow: Dataflow-Oriented Reinforcement Learning for Agentic LLMs","version":1},"cited_work":{"arxiv_id":"2605.15565","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.15565","snapshot_observed_at":"2026-08-16T00:15:04.848962Z","title":"AstraFlow: Dataflow-Oriented Reinforcement Learning for Agentic LLMs","venue":"cs.LG","work_id":"8938dfb9-bd8b-4266-a78e-2f95c3c7df42","year":2026},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.739486Z"},"links":{"cited_paper":"/paper/2605.15565","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:bee8b981b5d1021db2d4e1b77341bdf883385a22d450bc29603c3d9242daa8d8","observation_id":"14b14b77-d4f0-4ad5-ace9-8ac3a8523a23","resolution":{"observed_at":"2026-08-16T00:15:04.855445Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.13961","last_updated":"2026-04-14T15:12:44Z","snapshot_observed_at":"2026-08-16T16:18:35.731432Z","submitted_at":"2025-12-15T23:41:48Z","title":"Olmo 3","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.13961","snapshot_observed_at":"2026-08-16T00:15:04.747165Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.747165Z"},"links":{"cited_paper":"/paper/2512.13961","citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:b69b66700a03f347ae4c475cd29843a0c48f792f2c56a31e424ba90ce86e0f5b","observation_id":"5ca3e580-9a2d-456c-8bdd-442861246c0c","resolution":{"observed_at":"2026-08-16T00:15:04.747165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.829868Z","title":"SPIRAL [ 25], SPICE [31], Absolute Zero [47]): one model serves both roles with role-specific loss masks","venue":null,"work_id":"68070a13-76cd-4a1e-ae33-cab3b4e998fa","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.752437Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:89e768d3de60f6db083eae1e5ccca72ea22e71aed04ef2eb3500a489e3dc1f3c","observation_id":"a5d1f561-7b71-410b-b4d5-0a4db8808385","resolution":{"observed_at":"2026-08-16T00:15:06.833814Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.817820Z","title":null,"venue":null,"work_id":"8a09d344-17f5-4d21-a655-cd109c3e3952","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.757120Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:2e34315ed290ab2c971cc6165a4c21ef21f37cfc957186ede65d9982f2a72920","observation_id":"35a6214d-db4a-424a-9e49-e27de0c33ac4","resolution":{"observed_at":"2026-08-16T00:15:06.821870Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.805427Z","title":"Updated?","venue":null,"work_id":"198caac7-f1e8-412c-a2b8-5bad852589dc","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.761430Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:2212f13e8fa3409d4b657591d475abab357b6ffdb9c84ce18d483627099d5f7e","observation_id":"638d7632-dfa5-486d-a1a7-6542aa1c841d","resolution":{"observed_at":"2026-08-16T00:15:06.809353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.793483Z","title":null,"venue":null,"work_id":"c89f42d6-b3ed-4257-96fe-cefbc20379fd","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.766075Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:532b805e6bea8392253ebbb649896c980ed9456e59f767493e54594cff21a929","observation_id":"46274806-3801-4e2e-85fa-08471e40c932","resolution":{"observed_at":"2026-08-16T00:15:06.797513Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.781553Z","title":null,"venue":null,"work_id":"15bf6dec-c26c-4eab-a8d0-a73fdbe9cf86","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.770435Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:6aa47ed3fdcdd951b9d6b6f0a42d0568477b8669b4e73f2423f3e32f1c633801","observation_id":"f161cbfc-21f5-4bc0-aa22-439f28c13bee","resolution":{"observed_at":"2026-08-16T00:15:06.785421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.767048Z","title":null,"venue":null,"work_id":"229e280b-5d4b-4756-8900-11bbe5223bcf","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.775866Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:0d31052111cabc37b89ce971afa8820d44e830ecf40d4f0018b1aa15bc2875ab","observation_id":"494a49c9-ada6-4e4a-9c7d-ef6d159a6d21","resolution":{"observed_at":"2026-08-16T00:15:06.772211Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:06.752521Z","title":"responses","venue":null,"work_id":"4d3c32f7-690f-47ce-a1a9-26ea0febbec4","year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.780467Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:26b87f62a2a328a930dc9269720d025731dd2f7e239c38f2ce6e803a33fafbee","observation_id":"ba90d65d-a7c0-45aa-bcbe-c085f99c2fce","resolution":{"observed_at":"2026-08-16T00:15:06.757652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T00:15:04.495639Z","title":"arXiv:2509.19736 [cs]","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-16T00:15:04.495639Z"},"links":{"citing_paper":"/paper/2608.12253"},"observation_digest":"sha256:75c4b5c5ee2b121fe52468b2da1dcabdffa001ea9a30c95de0ae2ba3b6692637","observation_id":"39f02e0f-dbb2-40b5-ae65-3044662a48e9","resolution":{"observed_at":"2026-08-16T00:15:04.495639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.12253","last_updated":"2026-08-12T16:55:50Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T00:09:02.527975Z","submitted_at":"2026-08-12T16:55:50Z","title":"One Frozen Simulator Is Not Enough: Simulator Collapse in Multi-Agent RL"},"reference_resolution":{"displayed":75,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":1,"unresolved":59,"verified_exact":2,"verified_fuzzy":12},"total_outbound_references":75},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 75 of 75 outbound references and 0 inbound Pith citation observations for arXiv:2608.12253."}