{"as_of":"2026-08-16T14:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:725a40965c7d2305aaa22b35430b97f174bbdf040aebcd41359457d17148f0d6","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T14:56:00.329027Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":51,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:50:51.026135Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2504.02181","last_updated":"2026-04-22T01:49:17Z","snapshot_observed_at":"2026-07-30T09:24:14.725185Z","submitted_at":"2025-04-02T23:51:27Z","title":"A Survey of Scaling in Large Language Model Reasoning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T21:20:07.238992Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2504.02181"},"observation_digest":"sha256:1e9cffe16df439cb4e216db838fb8f50a1c725b8a95d18760a9e5f704a0307e0","observation_id":"e4bb1187-2562-46e3-9dc3-335f6fe88efa","resolution":{"observed_at":"2026-05-22T21:22:09.374887Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2505.10978","last_updated":"2025-10-28T15:11:36Z","snapshot_observed_at":"2026-07-29T19:20:21.974239Z","submitted_at":"2025-05-16T08:26:59Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-11T09:15:08.193357Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2505.10978"},"observation_digest":"sha256:02e0c4c555dc767824d638c36f1225833a404d423e307a7f3e7dacd848b447a1","observation_id":"8e668ac4-ae4d-4216-b8b9-cc1bdf5cac79","resolution":{"observed_at":"2026-05-11T09:15:08.493759Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2506.03610","last_updated":"2026-04-14T22:54:06Z","snapshot_observed_at":"2026-08-14T14:56:14.272480Z","submitted_at":"2025-06-04T06:40:33Z","title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-19T12:01:42.681135Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2506.03610"},"observation_digest":"sha256:eff19afa6847113ea71ac766fe3ffa1cba489ef418d8986b537faefff1a1e7bb","observation_id":"0d667442-84a5-4408-9b81-d21b07ebe5a5","resolution":{"observed_at":"2026-05-19T12:02:16.633951Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-07T10:30:34.130495Z","title":"Reinforcement learning for long-horizon interactive llm agents.arXiv preprint arXiv:2502.01600, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05333","last_updated":"2025-06-20T01:25:25Z","snapshot_observed_at":"2026-08-07T10:18:59.977399Z","submitted_at":"2025-06-05T17:59:24Z","title":"Kinetics: Rethinking Test-Time Scaling Laws","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T10:30:34.130495Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2506.05333"},"observation_digest":"sha256:bbb4bfca268bd2fbbcbeb1eeb0ccdbc6425d0cb22c9f0edf7a8e070bb5f2e4be","observation_id":"e8e64e95-ca1d-49b7-bed5-f699951ef985","resolution":{"observed_at":"2026-08-07T10:30:34.130495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2509.24948","last_updated":"2026-04-27T05:41:00Z","snapshot_observed_at":"2026-08-16T12:43:58.343317Z","submitted_at":"2025-09-29T15:45:19Z","title":"World-Env: Leveraging World Model as a Virtual Environment for VLA Post-Training","version":6},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-18T12:48:32.123998Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2509.24948"},"observation_digest":"sha256:66ce83f587acf70d253a37c2b0410d417c9f1608a966d3a0cbb6786f12391b66","observation_id":"0b5bd41e-3c99-44ec-a815-beaad8d3931f","resolution":{"observed_at":"2026-05-18T12:51:23.567751Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-03T00:13:25.232880Z","title":"Reinforce- ment learning for long-horizon interactive llm agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-13T00:03:13.729408Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.232880Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:bb66ac9d600f644ec8c553a7f660d05948fd843f85fa1c6bded982e3eec1f29c","observation_id":"925a5ce1-d2d7-4178-bfe5-2fb732e95e7c","resolution":{"observed_at":"2026-08-03T00:13:25.232880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.03675","last_updated":"2026-05-23T06:41:17Z","snapshot_observed_at":"2026-08-16T03:38:45.711568Z","submitted_at":"2026-04-04T10:23:46Z","title":"OASES: Outcome-Aligned Search-Evaluation Co-Training for Agentic Search","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-13T17:16:09.927267Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.03675"},"observation_digest":"sha256:b7cc985e389fddd84127d1c7ea4ea89e42298639c09fe002ae76afe1a9d2a461","observation_id":"ad7c6cd3-d20e-4af2-b1dd-39bbcd235b9c","resolution":{"observed_at":"2026-05-13T17:16:39.680321Z","resolver_source":"orphan_title_repair","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.07341","last_updated":"2026-08-12T17:33:31Z","snapshot_observed_at":"2026-08-16T08:37:09.555049Z","submitted_at":"2026-04-08T17:54:08Z","title":"ReCodeAgent: A Multi-agent Workflow for Language-Agnostic Translation and Validation of Large-Scale Repositories","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T17:29:37.642356Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.07341"},"observation_digest":"sha256:81d01da0b83fdafc754cbbfe777f12af09f3c9195b17f77d5b88346ff697835a","observation_id":"512c6297-bb55-4f2d-bdfc-82ef2af826de","resolution":{"observed_at":"2026-05-11T06:41:36.606515Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.07645","last_updated":"2026-04-08T23:11:12Z","snapshot_observed_at":"2026-08-15T05:23:39.352442Z","submitted_at":"2026-04-08T23:11:12Z","title":"PRIME: Training Free Proactive Reasoning via Iterative Memory Evolution for User-Centric Agent","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T17:17:59.156121Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.07645"},"observation_digest":"sha256:c05ed0c03ee6982b0940f1483fed99c328582ce4ad2f145e6167a557e929b809","observation_id":"ccb1273e-df61-4d0b-9635-b572b7a75b4d","resolution":{"observed_at":"2026-05-11T07:06:09.596177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.10674","last_updated":"2026-04-12T14:57:52Z","snapshot_observed_at":"2026-08-14T14:15:55.861698Z","submitted_at":"2026-04-12T14:57:52Z","title":"Skill-SD: Skill-Conditioned Self-Distillation for Multi-turn LLM Agents","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-10T15:28:07.981488Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.10674"},"observation_digest":"sha256:aced485668d28f3151a8f620e3ea3a7da6bb6378889139508421d837b9f61849","observation_id":"7feaa6e3-7208-4040-9399-24d247f613d7","resolution":{"observed_at":"2026-05-11T10:31:01.132467Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.24977","last_updated":"2026-04-27T20:22:42Z","snapshot_observed_at":"2026-08-14T09:15:57.722099Z","submitted_at":"2026-04-27T20:22:42Z","title":"A Survey on LLM-based Conversational User Simulation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T03:21:09.118243Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.24977"},"observation_digest":"sha256:b882463ef79a84628a372d1e01ffce6c356ca7e798fd27a6fcb7873c150ee1e2","observation_id":"3725c6d8-76ba-44c5-95b1-bf18b55c5d5c","resolution":{"observed_at":"2026-05-11T22:11:11.955831Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.26733","last_updated":"2026-05-15T12:03:42Z","snapshot_observed_at":"2026-07-06T23:12:19.460065Z","submitted_at":"2026-04-29T14:34:45Z","title":"FutureWorld: A Live Reinforcement Learning Environment for Predictive Agents with Real-World Outcome Rewards","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-07T11:35:24.834237Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.26733"},"observation_digest":"sha256:6c52e97d2d4a4c0c004a6190ccbf57ac3d61ad301ff52ba4c6c5a7b70697e347","observation_id":"e020dca7-2282-48be-b8af-941aec689ae8","resolution":{"observed_at":"2026-05-12T09:16:27.727101Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.26733","last_updated":"2026-05-15T12:03:42Z","snapshot_observed_at":"2026-07-06T23:12:19.460065Z","submitted_at":"2026-04-29T14:34:45Z","title":"FutureWorld: A Live Reinforcement Learning Environment for Predictive Agents with Real-World Outcome Rewards","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T03:13:17.539820Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.26733"},"observation_digest":"sha256:fe3752405b50e417ab093d9a541869586dc5d36b69c2108167bc41884bd8dcb2","observation_id":"210dde5e-a747-4990-bdaa-1517e3848351","resolution":{"observed_at":"2026-05-11T22:11:16.399865Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.26733","last_updated":"2026-05-15T12:03:42Z","snapshot_observed_at":"2026-07-06T23:12:19.460065Z","submitted_at":"2026-04-29T14:34:45Z","title":"FutureWorld: A Live Reinforcement Learning Environment for Predictive Agents with Real-World Outcome Rewards","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T01:45:31.691494Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.26733"},"observation_digest":"sha256:0ff009a093a5cca6da96ed0d37f7f77be1b278370d0351bad2380473e434b10a","observation_id":"3271b7ac-b713-4635-852e-de953a3b9536","resolution":{"observed_at":"2026-05-11T01:45:50.901886Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2604.26733","last_updated":"2026-05-15T12:03:42Z","snapshot_observed_at":"2026-07-06T23:12:19.460065Z","submitted_at":"2026-04-29T14:34:45Z","title":"FutureWorld: A Live Reinforcement Learning Environment for Predictive Agents with Real-World Outcome Rewards","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-19T17:20:07.747499Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2604.26733"},"observation_digest":"sha256:e5b773bf652d2097d6800ffb6aaaefcdbcca70dd538fff1ba4074c55c6f625b3","observation_id":"25088d7d-5bec-46f7-bc20-57dbb390f5d3","resolution":{"observed_at":"2026-05-19T17:22:42.017929Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.00425","last_updated":"2026-05-08T06:22:47Z","snapshot_observed_at":"2026-08-11T18:40:48.709904Z","submitted_at":"2026-05-01T05:54:37Z","title":"AEM: Adaptive Entropy Modulation for Multi-Turn Agentic Reinforcement Learning","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-09T19:56:27.170349Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.00425"},"observation_digest":"sha256:40bf9bfb4d11806d0543f6b2bf218b7f048d4cd0cf6c72e7b7b5f7018647d38e","observation_id":"ec1bfbce-90ef-4b2a-b5d2-a13ec96ed22e","resolution":{"observed_at":"2026-05-11T15:26:17.815362Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.00425","last_updated":"2026-05-08T06:22:47Z","snapshot_observed_at":"2026-08-11T18:40:48.709904Z","submitted_at":"2026-05-01T05:54:37Z","title":"AEM: Adaptive Entropy Modulation for Multi-Turn Agentic Reinforcement Learning","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-11T00:58:25.685484Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.00425"},"observation_digest":"sha256:d75a02832844f229f65e322cd98fc84990334423c3eaa728d2c6acb2c70c9183","observation_id":"7f592c01-f6c7-43d2-a313-fd7547883156","resolution":{"observed_at":"2026-05-11T04:55:59.108341Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.02572","last_updated":"2026-05-04T13:25:05Z","snapshot_observed_at":"2026-08-11T14:22:19.633592Z","submitted_at":"2026-05-04T13:25:05Z","title":"On Training Large Language Models for Long-Horizon Tasks: An Empirical Study of Horizon Length","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-08T18:13:25.735085Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.02572"},"observation_digest":"sha256:14520c469f50e03a79487d850ed3411b457519ec90ed90cc2d91d4cbe45947e3","observation_id":"150cb53b-4bd5-449b-950c-265fbaa860c0","resolution":{"observed_at":"2026-05-09T06:40:40.599054Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.06130","last_updated":"2026-05-12T10:25:23Z","snapshot_observed_at":"2026-08-15T04:46:13.708162Z","submitted_at":"2026-05-07T12:33:30Z","title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-08T10:23:52.522238Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.06130"},"observation_digest":"sha256:60127e068bc4b2ca8011a81eb6915f8333826c3041aae17ef8ceaee7e045fe0a","observation_id":"0c6d15d9-6bb8-41dc-a7cf-8dc0bfac18cc","resolution":{"observed_at":"2026-05-11T20:06:08.634344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.06130","last_updated":"2026-05-12T10:25:23Z","snapshot_observed_at":"2026-08-15T04:46:13.708162Z","submitted_at":"2026-05-07T12:33:30Z","title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-11T02:00:00.663355Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.06130"},"observation_digest":"sha256:906af3f2ffa793edae266c0805ca616691b8dbc44ae9d605317e1ea0c94b4cf1","observation_id":"8c63fbe8-01a5-46d0-b1b2-d810a5660313","resolution":{"observed_at":"2026-05-11T04:05:55.145588Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.06130","last_updated":"2026-05-12T10:25:23Z","snapshot_observed_at":"2026-08-15T04:46:13.708162Z","submitted_at":"2026-05-07T12:33:30Z","title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","version":3},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-13T07:17:13.708752Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.06130"},"observation_digest":"sha256:2764196f4c2259a128df94f34eb666871f6ad5eb3d228b0298b198f59395b455","observation_id":"fc4b1324-6e69-4bb9-b185-37d0cbaf4b7b","resolution":{"observed_at":"2026-05-13T07:17:28.298280Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.06957","last_updated":"2026-05-07T21:22:33Z","snapshot_observed_at":"2026-07-06T23:19:20.674889Z","submitted_at":"2026-05-07T21:22:33Z","title":"Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T01:10:01.008358Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.06957"},"observation_digest":"sha256:03773a9c5a5aabeb84fe35768684262d388411ae4b3933f5cde3f32bbd2782a0","observation_id":"5002871f-5011-490c-8051-83c5be015592","resolution":{"observed_at":"2026-05-11T04:40:59.236958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.07462","last_updated":"2026-05-08T09:10:17Z","snapshot_observed_at":"2026-08-11T16:49:22.179421Z","submitted_at":"2026-05-08T09:10:17Z","title":"The Moltbook Files: A Harmless Slopocalypse or Humanity's Last Experiment","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-11T01:54:49.131461Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.07462"},"observation_digest":"sha256:7b9707b94e18cf8b8ffc76e2841012a255f090a08411472bebe6506175e52964","observation_id":"70cf5383-fd29-4172-9093-e0c0ee7f2ffb","resolution":{"observed_at":"2026-05-11T01:55:51.066392Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.07725","last_updated":"2026-08-03T03:33:59Z","snapshot_observed_at":"2026-08-14T23:07:50.063910Z","submitted_at":"2026-05-08T13:30:42Z","title":"SOD: Step-wise On-policy Distillation for Small Language Model Agents","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-11T02:25:59.056181Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.07725"},"observation_digest":"sha256:26a37cdaf2bacbb67c4d9f797d9f32c8e679fae7823ccec82df10fa9db4a51fc","observation_id":"f4304d3e-0096-4a0d-9b56-6c8036d8642a","resolution":{"observed_at":"2026-05-11T03:40:54.478495Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-04T05:20:45.157089Z","title":"Reinforcement learning for long-horizon interactive llm agents.arXiv preprint arXiv:2502.01600, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.07725","last_updated":"2026-08-03T03:33:59Z","snapshot_observed_at":"2026-08-14T23:07:50.063910Z","submitted_at":"2026-05-08T13:30:42Z","title":"SOD: Step-wise On-policy Distillation for Small Language Model Agents","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-04T05:20:45.157089Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.07725"},"observation_digest":"sha256:6da84c742e152951b9e21f8c6f88ce2f54ad9190a9c14334acd8170e6fb733d8","observation_id":"9f7caa68-52f5-452f-99dc-cf86f58209e2","resolution":{"observed_at":"2026-08-04T05:20:45.157089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.14558","last_updated":"2026-05-14T08:33:02Z","snapshot_observed_at":"2026-07-06T23:25:58.895612Z","submitted_at":"2026-05-14T08:33:02Z","title":"Resolving Action Bottleneck: Agentic Reinforcement Learning Informed by Token-Level Energy","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T01:46:24.724553Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.14558"},"observation_digest":"sha256:0d2804e008fd48c2ffd44b9450f896e3c2fb090f504d2b1436a0904fdbb348af","observation_id":"8d01e1ee-1956-4538-94b6-9d60d3da26f7","resolution":{"observed_at":"2026-05-15T01:48:28.541838Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.19762","last_updated":"2026-05-19T12:37:01Z","snapshot_observed_at":"2026-08-15T01:54:37.990087Z","submitted_at":"2026-05-19T12:37:01Z","title":"What Really Improves Mathematical Reasoning: Structured Reasoning Signals Beyond Pure Code","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-20T05:06:46.360174Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.19762"},"observation_digest":"sha256:1ec0ea046cd708fbc2029a75973e552d8c2c5f265896ab4cbd7f404667f84f94","observation_id":"225e3096-042b-4d89-9c8e-8faa4995535c","resolution":{"observed_at":"2026-05-20T05:08:05.103512Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.20061","last_updated":"2026-05-19T16:19:29Z","snapshot_observed_at":"2026-08-15T18:55:46.396549Z","submitted_at":"2026-05-19T16:19:29Z","title":"Rewarding Beliefs, Not Actions: Consistency-Guided Credit Assignment for Long-Horizon Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T05:35:45.084011Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.20061"},"observation_digest":"sha256:7e66b35e23e40fffd910d37b7020aecfdaaec8ef3c49c60068670cace033bbc2","observation_id":"fe5798ea-0ed8-40ee-bd67-1950eebb2d83","resolution":{"observed_at":"2026-05-20T05:38:05.569721Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2605.25384","last_updated":"2026-05-25T03:21:12Z","snapshot_observed_at":"2026-07-06T23:35:21.734804Z","submitted_at":"2026-05-25T03:21:12Z","title":"GeoMathCode: Understanding Interleaved Math-Code Reasoning for Geometry Problem Solving","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-29T22:44:39.024780Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2605.25384"},"observation_digest":"sha256:ff96cbd57294e026d7ae1f7df5cd247c989eadc24f6f0d5082ec8e9d287a4cef","observation_id":"32850909-3bfb-461d-8a2e-bb147a226f0c","resolution":{"observed_at":"2026-06-29T22:54:01.558000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.02355","last_updated":"2026-06-01T15:02:59Z","snapshot_observed_at":"2026-08-16T00:03:12.293319Z","submitted_at":"2026-06-01T15:02:59Z","title":"SIRI: Self-Internalizing Reinforcement Learning with Intrinsic Skills for LLM Agent Training","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-06-28T14:33:00.408984Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.02355"},"observation_digest":"sha256:98e89042567725e460e2fff035fe4363dcd9f5d4b9b156e9abebd0a38273e5a7","observation_id":"2e23d26e-160b-4cee-8734-d8aaf6893079","resolution":{"observed_at":"2026-07-01T23:16:23.914071Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.03108","last_updated":"2026-06-02T03:47:48Z","snapshot_observed_at":"2026-08-15T15:01:56.287299Z","submitted_at":"2026-06-02T03:47:48Z","title":"EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-06-28T10:30:42.057301Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.03108"},"observation_digest":"sha256:add18fa8e8597b76ce782b5adb8c532df24a58129f65bfe92e5816b5ff4764be","observation_id":"eabaf675-ae68-45a3-9704-648abfd971f8","resolution":{"observed_at":"2026-07-02T02:56:29.163648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.05885","last_updated":"2026-06-04T08:54:09Z","snapshot_observed_at":"2026-08-08T00:56:45.958148Z","submitted_at":"2026-06-04T08:54:09Z","title":"When Denser Credit Is Not Enough: Evidence-Calibrated Policy Optimization for Long-Horizon LLM Agent Training","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-28T02:17:32.324432Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.05885"},"observation_digest":"sha256:c520875ca33853dc4288202f35bb646fa739dc281babcf5db404fb510199f2b0","observation_id":"f2f634f9-c348-4780-b013-cf731755b35d","resolution":{"observed_at":"2026-07-02T12:16:56.938852Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.08867","last_updated":"2026-06-07T22:44:00Z","snapshot_observed_at":"2026-08-03T19:07:50.944087Z","submitted_at":"2026-06-07T22:44:00Z","title":"Building Customer Support AI Agents at 100M-User Scale: An Evaluation-Driven Framework","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T18:21:57.096578Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.08867"},"observation_digest":"sha256:b3059bade2480f059b3026a188c1070a75910d7ebba7f2f17cb5e27ce1de58ed","observation_id":"32262e4f-cb6c-4ad6-8590-788e32e41381","resolution":{"observed_at":"2026-07-02T23:17:29.488885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:2d1e199f94c2a3963da0c9745738a0b785dafacb6204e4b40e5e67315b504478","observation_id":"ca03c586-95e5-467b-a080-52bc8ff08e3b","resolution":{"observed_at":"2026-07-03T21:18:58.371419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.22995","last_updated":"2026-06-22T08:12:47Z","snapshot_observed_at":"2026-08-13T02:10:15.486582Z","submitted_at":"2026-06-22T08:12:47Z","title":"Group-Graph Policy Optimization for Long-Horizon Agentic Reinforcement Learning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T08:44:23.085858Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.22995"},"observation_digest":"sha256:bc49c245f4560073b342cce4e94b5e5954523f3113b0aad3182a509af09de4f9","observation_id":"5972265b-c31e-475e-bb50-4ecf7eb7d38e","resolution":{"observed_at":"2026-07-04T10:29:45.778295Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.26918","last_updated":"2026-06-25T11:53:41Z","snapshot_observed_at":"2026-08-12T12:41:32.972934Z","submitted_at":"2026-06-25T11:53:41Z","title":"Diagnosing Task Insensitivity in Language Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T04:58:11.929896Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.26918"},"observation_digest":"sha256:55125b989f259fbfa84403a2b20a1cea3c7e7b0603062236a9b752058a22a78a","observation_id":"151b103e-0bf4-4ebc-a0a6-affd7a86fcfa","resolution":{"observed_at":"2026-07-04T13:39:51.584015Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.31504","last_updated":"2026-06-30T11:22:24Z","snapshot_observed_at":"2026-08-12T18:44:43.168315Z","submitted_at":"2026-06-30T11:22:24Z","title":"SimpleSearch-VL: A Simple Recipe for Multimodal Agentic Deep Search","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-07-01T06:02:48.532478Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.31504"},"observation_digest":"sha256:d22df44c95765aab1f9d2aaa0992488a6a632d0667492722be448421d94ee13c","observation_id":"b17426f0-44c3-46f9-a691-930410cb760e","resolution":{"observed_at":"2026-07-01T09:55:41.045805Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2607.01897","last_updated":"2026-07-02T08:50:32Z","snapshot_observed_at":"2026-08-13T06:18:38.670996Z","submitted_at":"2026-07-02T08:50:32Z","title":"Rank-Then-Act: Reward-Free Control from Frame-Order Progress","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-03T17:42:44.309030Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2607.01897"},"observation_digest":"sha256:06d3a77c4648d9f3a04dd97340b63411f70efe1aa474378005c0e5167b54cc3d","observation_id":"c5c2b98d-1af9-4e9c-b0eb-147293a4d3f2","resolution":{"observed_at":"2026-07-03T17:48:45.215284Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2607.02431","last_updated":"2026-07-02T17:00:37Z","snapshot_observed_at":"2026-08-15T17:59:19.515738Z","submitted_at":"2026-07-02T17:00:37Z","title":"WorldSample: Closed-loop Real-robot RL with World Modelling","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-03T10:57:40.128651Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2607.02431"},"observation_digest":"sha256:8928e5bc321979f642ce9b6685350c9c81b9fee8dcb6e1de8052bca87e5dd985","observation_id":"07643e0c-ac7f-4345-8fa3-6f137fc0a8db","resolution":{"observed_at":"2026-07-03T10:58:02.459374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-07-12T05:22:49.815699Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.03025","last_updated":"2026-07-03T07:12:44Z","snapshot_observed_at":"2026-08-15T03:22:35.940196Z","submitted_at":"2026-07-03T07:12:44Z","title":"Human-Centric Reflective Architecture for Human-AI Collaborative Decision-Making","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-12T05:22:49.815699Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2607.03025"},"observation_digest":"sha256:ebc3965f1c5819426dd594a6ae337f98e899c4ee52167e93767e19304db29e45","observation_id":"1060e7b5-7a1f-44ad-ae9b-224b1b3a6971","resolution":{"observed_at":"2026-07-12T05:22:49.815699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-07-11T10:50:54.419477Z","title":"arXiv preprint arXiv:2502.01600 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04963","last_updated":"2026-07-06T11:50:06Z","snapshot_observed_at":"2026-08-13T15:58:56.004748Z","submitted_at":"2026-07-06T11:50:06Z","title":"STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-07-11T10:50:54.419477Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2607.04963"},"observation_digest":"sha256:a84ff4b18e54cd3b2c94485f333e6b3473de0cbf129ab1ade9ec4a83bfe401d3","observation_id":"b644b619-26d8-48ea-91df-0d76deedb835","resolution":{"observed_at":"2026-07-11T10:50:54.419477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-01T01:29:10.250529Z","title":"Reinforcement Learning for Long-Horizon Interactive","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25816","last_updated":"2026-07-28T15:00:10Z","snapshot_observed_at":"2026-08-13T20:15:29.585970Z","submitted_at":"2026-07-28T15:00:10Z","title":"Speculate While You Reason: Teaching Agents to Predict Their Next Tool Call via Joint Agent-Speculator RL","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-01T01:29:10.250529Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2607.25816"},"observation_digest":"sha256:0e69b10cc3486b1c6086a73112794519f38c9098e82d5d906c01abab7e3ef642","observation_id":"5fa8e51b-6044-4349-84d4-f8d597be9f87","resolution":{"observed_at":"2026-08-01T01:29:10.250529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-04T23:23:14.629933Z","title":"Reinforcement learning for long-horizon interactive llm agents.arXiv preprint arXiv:2502.01600, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01667","last_updated":"2026-08-03T04:01:36Z","snapshot_observed_at":"2026-08-09T13:33:19.122605Z","submitted_at":"2026-08-03T04:01:36Z","title":"TCPO: Turn-Level Credit Policy Optimization","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T23:23:14.629933Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.01667"},"observation_digest":"sha256:627ade84928dbb3459ce242df7c49624a26596284d4d65859578c050a81fc81a","observation_id":"74360757-7012-4b6a-9243-770addcc35bc","resolution":{"observed_at":"2026-08-04T23:23:14.629933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T23:15:50.724043Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.03223","last_updated":"2026-08-04T06:56:47Z","snapshot_observed_at":"2026-08-15T06:09:27.350327Z","submitted_at":"2026-08-04T06:56:47Z","title":"Agentic Reinforcement Learning with Self-Distilled Reward Shaping","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T23:15:50.724043Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.03223"},"observation_digest":"sha256:f06745a91325bc3e6c66169aa9dbab78e42850a779d54a6a5918ab7a88918270","observation_id":"9adefe45-cc4b-402a-a50d-2d780189b19a","resolution":{"observed_at":"2026-08-05T23:15:50.724043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T17:56:49.695690Z","title":"Reinforcement Learning for Long-Horizon Interactive","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03499","last_updated":"2026-08-04T11:42:26Z","snapshot_observed_at":"2026-08-14T05:40:05.736283Z","submitted_at":"2026-08-04T11:42:26Z","title":"WeClawArena: An Auditable Sandbox and Benchmark for Cross-User Agents Collaboration and Security in Human-Centered Agent Networks","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T17:56:49.695690Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.03499"},"observation_digest":"sha256:dea3467d36e2761ce660f8f730e9365335e3a19f4e99aa6074cda828a1e26fdc","observation_id":"52bf7d1e-7a78-4cbc-9950-ef6e8145ea06","resolution":{"observed_at":"2026-08-05T17:56:49.695690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-10T04:30:27.019408Z","title":"arXiv preprint arXiv:2502.01600 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.06410","last_updated":"2026-08-03T22:50:49Z","snapshot_observed_at":"2026-08-14T20:57:12.837382Z","submitted_at":"2026-08-03T22:50:49Z","title":"ADIAS: Automated Design of Interactive Agentic Systems","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-10T04:30:27.019408Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.06410"},"observation_digest":"sha256:43504bf743f8588cdeecea9da86c1fdfd1758fa4db6abb7b5611ead993213be2","observation_id":"f02e847f-6de2-4d70-9e96-b18755c4a78b","resolution":{"observed_at":"2026-08-10T04:30:27.019408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-10T19:29:14.787945Z","title":"a henb \\","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.06861","last_updated":"2026-08-07T06:38:12Z","snapshot_observed_at":"2026-08-15T19:16:11.258131Z","submitted_at":"2026-08-07T06:38:12Z","title":"Gated-BEPO: Confidence-Gated Bellman Credit Assignment for Large Language Model Agents","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-10T19:29:14.787945Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.06861"},"observation_digest":"sha256:98c9cec6d23e53e3c13be834ae9e09def2249fd0025d80de40e22a1a79934a4a","observation_id":"a1b4e244-6b66-4eee-8d81-a7f6395fbe2e","resolution":{"observed_at":"2026-08-10T19:29:14.787945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-10T14:00:52.751101Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.07147","last_updated":"2026-08-07T12:07:12Z","snapshot_observed_at":"2026-08-15T02:43:35.087716Z","submitted_at":"2026-08-07T12:07:12Z","title":"DiDPO: Diff-in-Diff Policy Optimization for Coding Agent Training","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T14:00:52.751101Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.07147"},"observation_digest":"sha256:9939e8853dbed09aeb5eda375eb3810ec2783e12394152857df364cdffe28245","observation_id":"604af2c4-39c3-4a80-8fb9-b61d552f1bdd","resolution":{"observed_at":"2026-08-10T14:00:52.751101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-12T00:50:51.026135Z","title":"arXiv preprint arXiv:2502.01600 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.07855","last_updated":"2026-08-08T01:50:55Z","snapshot_observed_at":"2026-08-13T23:21:37.017678Z","submitted_at":"2026-08-08T01:50:55Z","title":"CommitKV: Lifecycle-Aware KV Cache Compression via Commit Transitions for Multi-Turn Agents","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-12T00:50:51.026135Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.07855"},"observation_digest":"sha256:1dce4e96ae5fcd51d2043b58f763b20c533263638908ae1030c6c02368a6cdb6","observation_id":"e12b3ee5-81bf-4fa0-b109-8e6a4395e104","resolution":{"observed_at":"2026-08-12T00:50:51.026135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-12T00:20:30.881986Z","title":"Test-time adaptation for llm agents via environment interaction","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08255","last_updated":"2026-08-08T17:32:34Z","snapshot_observed_at":"2026-08-15T14:42:29.896951Z","submitted_at":"2026-08-08T17:32:34Z","title":"Learning from Environmental Feedback: Credit Assignment across Multiple Timescales for Agentic Reinforcement Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T00:20:30.881986Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.08255"},"observation_digest":"sha256:a4a8c45ee5bfce796230ba92115193c745edb24f4481b100a5b14d64c74986fa","observation_id":"d90394ae-8bf4-421a-81a6-97acf69a6280","resolution":{"observed_at":"2026-08-12T00:20:30.881986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-11T15:14:31.186850Z","title":"a henb \\","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09555","last_updated":"2026-08-10T12:53:06Z","snapshot_observed_at":"2026-08-14T21:13:12.975026Z","submitted_at":"2026-08-10T12:53:06Z","title":"Bidirectional Context Self-Distillation for Reinforcement Learning of Skill-Based LLM Agents","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-11T15:14:31.186850Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2608.09555"},"observation_digest":"sha256:886b667c2cb8a1852e7475d798762c15e913856762c75753c7d2d7cd7d638ac7","observation_id":"7903429c-8806-47f5-980f-186ef96cb48c","resolution":{"observed_at":"2026-08-11T15:14:31.186850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.01600/citation-record","integrity":"/paper/2502.01600/integrity","json":"/paper/2502.01600/citation-record.json","paper":"/paper/2502.01600"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.175690Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.175690Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:3ed319a66da890e1899a6793241a106907f9f8a4f119c9371cf118ada608983d","observation_id":"be444522-3b84-4f00-a6d2-27129c6c827f","resolution":{"observed_at":"2026-08-09T14:56:00.175690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.926420Z","title":"Back to basics: Revisiting REINFORCE -style optimization for learning from human feedback in LLMs","venue":null,"work_id":"091d10b1-9ebe-48d1-99a3-7da5845af25c","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.180044Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:40048c1f0f978cb257f10e4b51383a4b2c0d9b64567b1f99ad389fb2bcde981b","observation_id":"9e9d7b82-4b1c-40d7-969d-86de32af08b9","resolution":{"observed_at":"2026-08-09T14:56:00.929673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.917688Z","title":"Thinking fast and slow with deep learning and tree search","venue":null,"work_id":"f9037542-2523-4c4c-a385-f79237ea5fec","year":2017},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.183508Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:34a6baa3a3be43e9d739af5880501325db5af2fbe434b333d80b8510a1e2bf54","observation_id":"9357bc52-4d7e-4185-964c-58cfaf9368c7","resolution":{"observed_at":"2026-08-09T14:56:00.920816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11896","last_updated":"2024-06-14T17:49:55Z","snapshot_observed_at":"2026-08-16T13:42:51.892765Z","submitted_at":"2024-06-14T17:49:55Z","title":"DigiRL: Training In-The-Wild Device-Control Agents with Autonomous Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11896","snapshot_observed_at":"2026-08-09T14:56:00.187032Z","title":"DigiRL : Training in-the-wild device-control agents with autonomous reinforcement learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.187032Z"},"links":{"cited_paper":"/paper/2406.11896","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:d324fa051353b1c6c4df003aa0f60aeacd10d14199c58da2185258342c3ef86d","observation_id":"3c600c5b-3e54-43f7-96d3-3ae9a1749549","resolution":{"observed_at":"2026-08-09T14:56:00.187032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.909228Z","title":"Grounding large language models in interactive environments with online reinforcement learning","venue":null,"work_id":"053279d4-f77d-4d48-85bf-f1e6724d5491","year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.190655Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:f57afabfc0a81b685bb68d268e528e4c0314d3efeb5139267b4b72457a1cfea1","observation_id":"122fa5cb-c3db-4f03-89ea-9df8ea0b834b","resolution":{"observed_at":"2026-08-09T14:56:00.912383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05915","last_updated":"2023-10-09T17:58:38Z","snapshot_observed_at":"2026-08-13T05:53:40.193848Z","submitted_at":"2023-10-09T17:58:38Z","title":"FireAct: Toward Language Agent Fine-tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05915","snapshot_observed_at":"2026-08-09T14:56:00.193783Z","title":"FireAct : Toward language agent fine-tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.193783Z"},"links":{"cited_paper":"/paper/2310.05915","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:2c5cc8ab53cd59daf24c34e60bf6bc3f9241b416e8ba4ec2a0f9fe0b59b1dbda","observation_id":"e48f99d2-9cf3-4d39-a050-74ea028dfc96","resolution":{"observed_at":"2026-08-09T14:56:00.193783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-09T14:56:00.197270Z","title":"DeepSeek-R1 : Incentivizing reasoning capability in LLMs via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.197270Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:58cf14ee03d073fa972b7ff093aed635911186bdd782649f93b467bd329a8840","observation_id":"97a8fec7-3e07-4384-8931-71b1345616b9","resolution":{"observed_at":"2026-08-09T14:56:00.197270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-09T14:56:00.200570Z","title":"The Llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.200570Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:f71f89e1de01354c4091efa85dbf923caf6d8f32b4a614e5537577f2bc35c4f9","observation_id":"b5796287-07ce-4448-a87b-b07710f8b6f5","resolution":{"observed_at":"2026-08-09T14:56:00.200570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.203577Z","title":"D., Oosterhuis, H., de Rijke, M., and Shukla, S","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.203577Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:7eac4ad7212ce649416df7a420ee3987827653941f6df14f77a225cd1ce34118","observation_id":"d841b603-7417-495e-86db-f2062e30d551","resolution":{"observed_at":"2026-08-09T14:56:00.203577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04642","last_updated":"2024-03-07T16:36:29Z","snapshot_observed_at":"2026-08-16T14:11:54.240619Z","submitted_at":"2024-03-07T16:36:29Z","title":"Teaching Large Language Models to Reason with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04642","snapshot_observed_at":"2026-08-09T14:56:00.206362Z","title":"C., Nalmpantis, C., Dwivedi-Yu, J., Zhuravinskyi, M., Hambro, E., Sukhbaatar, S., and Raileanu, R","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.206362Z"},"links":{"cited_paper":"/paper/2403.04642","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:0c9f9af9e9a9890f029d92f8085edd5b97548dd8c00655f19e05586f682db42b","observation_id":"5844cb03-48df-4753-95b5-fc08356f6552","resolution":{"observed_at":"2026-08-09T14:56:00.206362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.900146Z","title":"J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., and Chen, W","venue":null,"work_id":"539a3011-b880-48e3-b9cd-ce24b07c162f","year":2022},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.211006Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:677e0e7cf80c55a7dcb2c0529c3009fa26002be7ae9a2c6cd1118a5e4e7cad77","observation_id":"f3e6733f-79b7-4d4f-8ce6-dd4c2300cce0","resolution":{"observed_at":"2026-08-09T14:56:00.903345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.891639Z","title":"P., Littman, M","venue":null,"work_id":"b8660c74-758f-4e29-88f3-2e2be89a5eb2","year":1998},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.214221Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:2dec936025fe5e00e0937841a99c37cec6fa93856f5109b578196f111fe71964","observation_id":"20058a19-7f85-4784-97f4-44a1377fe38d","resolution":{"observed_at":"2026-08-09T14:56:00.894633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.217235Z","title":"and Langford, J","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.217235Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:9186dbe8f7874c428127e83cdb99e6d2d9760f68040414adcb03f5ecae0c943e","observation_id":"70bc154c-7af5-47f7-8fe4-2d74ec3217ac","resolution":{"observed_at":"2026-08-09T14:56:00.217235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01679","last_updated":"2025-06-03T20:51:06Z","snapshot_observed_at":"2026-08-16T13:13:18.052076Z","submitted_at":"2024-10-02T15:49:30Z","title":"VinePPO: Refining Credit Assignment in RL Training of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01679","snapshot_observed_at":"2026-08-09T14:56:00.220169Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.220169Z"},"links":{"cited_paper":"/paper/2410.01679","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:007738553abafa8288a723da8ac1ad5791ec5883bbc2f8302a5ee85dd439d9d5","observation_id":"d8060d98-83f4-4402-b46b-a9de15137134","resolution":{"observed_at":"2026-08-09T14:56:00.220169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.876937Z","title":"Language models can solve computer tasks","venue":null,"work_id":"29e4edb0-23ec-4a19-81cf-62236e5e9bc1","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.223411Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:c885a66f0ddc6d86e1f01a8a1570c36854c8bb57b4bf33315cf7043ec7e93c49","observation_id":"e1852615-2ed0-4e33-b45a-294be7c124c1","resolution":{"observed_at":"2026-08-09T14:56:00.880119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.868120Z","title":"Buy 4 reinforce samples, get a baseline for free! In ICLR 2019 Workshops, 2019","venue":null,"work_id":"2ee3ce59-2d2c-4e9c-9903-ed43a0aac95f","year":2019},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.226220Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:b0d67313e97bc6f4d715719930e153c0c9a9f9e9d67e0b928ec71596b907a3ae","observation_id":"c7027d52-7992-4140-a450-a787d6c6bd78","resolution":{"observed_at":"2026-08-09T14:56:00.871327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.859191Z","title":"H., Gonzalez, J., Zhang, H., and Stoica, I","venue":null,"work_id":"08547fc9-342f-447f-ab64-e359098870b8","year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.229213Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:23750deaae0d7d5628a7e612eab91b7461da8a37d4593f9d805a98e4dd83778f","observation_id":"f2c3648e-c89e-4fb3-8f93-2b4fcb7a5209","resolution":{"observed_at":"2026-08-09T14:56:00.862493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-08-10T16:05:13.426341Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-09T14:56:00.232188Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.232188Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:eca87855d8fc3a7a62e862d0aa87bef837ec16a4e9666f3adbc1e409c399b4c7","observation_id":"9b6f215a-b079-49ea-ac85-70693b68404e","resolution":{"observed_at":"2026-08-09T14:56:00.232188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03502","last_updated":"2024-07-03T21:01:12Z","snapshot_observed_at":"2026-08-16T13:37:14.651237Z","submitted_at":"2024-07-03T21:01:12Z","title":"AgentInstruct: Toward Generative Teaching with Agentic Flows","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03502","snapshot_observed_at":"2026-08-09T14:56:00.235295Z","title":"AgentInstruct : Toward generative teaching with agentic flows","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.235295Z"},"links":{"cited_paper":"/paper/2407.03502","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:f9649f7e9757bdbdcd6e57b5d90b086c2fad164422f54f853293825c4f174258","observation_id":"d7f30d4d-343c-439f-8a42-8433b614696b","resolution":{"observed_at":"2026-08-09T14:56:00.235295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-08-07T17:14:39.278754Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-09T14:56:00.238491Z","title":"WebGPT : Browser-assisted question-answering with human feedback","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.238491Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:0d11f1eb94118f94b1341f9380eaed78e0bc488b9358fc9c70f3bfd9264a1085","observation_id":"d447578b-626f-48f9-ae56-3057b45e943e","resolution":{"observed_at":"2026-08-09T14:56:00.238491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.850546Z","title":"D., and Barzilay, R","venue":null,"work_id":"79a0c5a0-b370-4490-8307-1422102ba88c","year":2015},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.241690Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:73e7d4538b9ca18e5713500ecf6f89850fbccd62a7485baec0574f33249f16be","observation_id":"1c93a203-4b43-4420-b096-0ee13fc603c2","resolution":{"observed_at":"2026-08-09T14:56:00.853634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.841699Z","title":"Introducing OpenAI o1, 2024","venue":null,"work_id":"c0374326-aeb5-4228-b03b-476aa7533980","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.244583Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:c38d845d1fe40ea3df835f626fed46ad09ebf1f3d0595a6b78128507504fb354","observation_id":"52c2be5b-19a6-4376-83ea-4ef330ac51b6","resolution":{"observed_at":"2026-08-09T14:56:00.844786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.247681Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.247681Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:b45ca7cc65487ca0176732fdd969acf5d258fe496b58922ad5360b81ef3793eb","observation_id":"ea2a00a5-9396-4569-91ee-7b72038ef238","resolution":{"observed_at":"2026-08-09T14:56:00.247681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07199","last_updated":"2024-08-13T20:52:13Z","snapshot_observed_at":"2026-08-14T15:46:52.927758Z","submitted_at":"2024-08-13T20:52:13Z","title":"Agent Q: Advanced Reasoning and Learning for Autonomous AI Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07199","snapshot_observed_at":"2026-08-09T14:56:00.250658Z","title":"Agent Q : Advanced reasoning and learning for autonomous AI agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.250658Z"},"links":{"cited_paper":"/paper/2408.07199","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:b25c19d5dc5e48cbce36709cc2a444c9960f3dabde2aaf68e321c936266b5241","observation_id":"832d4fbe-3772-4836-ab60-ac71b6e5e84d","resolution":{"observed_at":"2026-08-09T14:56:00.250658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.828159Z","title":"ToolLLM : Facilitating large language models to master 16000+ real-world APIs","venue":null,"work_id":"c3080b7b-051a-4334-a7bc-c2dba4546b79","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.253897Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:42eb2eb1e492e0b49ca73c098f779022b265c045aa154f5c1b53c2edec4d298f","observation_id":"3f99d275-8b4b-4e73-9702-44873ce220c8","resolution":{"observed_at":"2026-08-09T14:56:00.831141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.820272Z","title":"Toolformer: Language models can teach themselves to use tools","venue":null,"work_id":"60e22e5a-f86c-4fa7-aacb-5cd997eafdec","year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.256781Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:3854066124276112359b87f6e32d7aa54ffea52097c2abbdccb6776813d17f8d","observation_id":"ed9810bc-ea2f-4e26-a9c2-836040c6721a","resolution":{"observed_at":"2026-08-09T14:56:00.823270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.811864Z","title":"I., and Abbeel, P","venue":null,"work_id":"81c73ccc-7e2e-4da1-9d19-53ad3691174f","year":2016},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.260358Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:dcdd7b0f6714bd837f7154047616c2e24aee07faf8b775d0088602cc8f60e41b","observation_id":"25e64932-c570-4820-adec-0a67c5139792","resolution":{"observed_at":"2026-08-09T14:56:00.814874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-08-15T20:26:32.102285Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-09T14:56:00.263354Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.263354Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:ea4117aa71909efe48fcd310045f2c3c9223304d091d5647aa2c509833ca63bc","observation_id":"1a64858a-7356-4e9a-84af-064cb4b44654","resolution":{"observed_at":"2026-08-09T14:56:00.263354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-09T14:56:00.266419Z","title":"DeepSeekMath : Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.266419Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:08e332efb50ec5c2df72589e1fa4ae4c5caca2a6e9cdf3db8131881996409afa","observation_id":"a20db280-cd63-4995-9cba-3a91d9e32b3a","resolution":{"observed_at":"2026-08-09T14:56:00.266419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.803321Z","title":"Direct multi-turn preference optimization for language agents","venue":null,"work_id":"d6606a17-5ad1-4959-a711-b04b060a4d69","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.269423Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:8a4057372b327f2e7eb3790078c459097990849735177c12d3606067f7f11405","observation_id":"da300191-1603-4e48-ae49-8c0440154f38","resolution":{"observed_at":"2026-08-09T14:56:00.806541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.794601Z","title":"Reflexion: Language agents with verbal reinforcement learning","venue":null,"work_id":"7e649c93-d77d-4f92-808f-e51c8fd563c6","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.272361Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:3d453089cf5c5a8c0d3d8c3d88d20f0dc83ec86d3701530ec7a52d937dd15cde","observation_id":"264b9455-85ab-447e-9360-3ccafe36cfbe","resolution":{"observed_at":"2026-08-09T14:56:00.798035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.786402Z","title":"D., Agarwal, R., Anand, A., Patil, P., Garcia, X., Liu, P","venue":null,"work_id":"32c99661-9e1c-424b-ac19-8b121ba33849","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.275289Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:46c0b526ab8da80686154bff0bd8bb321a3b2bf0b9e1ca91b00a686d42172608","observation_id":"8fc187c8-522a-4390-81c4-0effd84dfc5c","resolution":{"observed_at":"2026-08-09T14:56:00.789415Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.777925Z","title":"M., Lowe, R., Voss, C., Radford, A., Amodei, D., and Christiano, P","venue":null,"work_id":"273654d1-2417-4dbc-add1-b55cc7be48a3","year":2020},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.278327Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:714b7d864f57418c7921139481f005d5a8f7de132911dced6736f17e8326e5de","observation_id":"a14132ce-3559-4151-93b0-a07490148853","resolution":{"observed_at":"2026-08-09T14:56:00.780972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.767533Z","title":"torchtune: PyTorch's finetuning library, April 2024","venue":null,"work_id":"3912b2c7-90bb-4746-bfd4-ee07b95ef3d6","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.281144Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:51fcdf3f09246fcd0ab4d0db5f06fd81abb6991d791530a9387e7cc11783cfd1","observation_id":"ad18a1a7-50cc-4f3f-8baa-34fb59e4d741","resolution":{"observed_at":"2026-08-09T14:56:00.770623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.759767Z","title":"A pp W orld: A controllable world of apps and people for benchmarking interactive coding agents","venue":null,"work_id":"e4f37b0c-946f-43f5-a208-45f386691adb","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.284213Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:6f93798d0c0941186e565c6d764e3efe0044f17d68c2359a4c4d0c1479def18d","observation_id":"7b3adcb7-ac58-4d3b-b959-4641a548575d","resolution":{"observed_at":"2026-08-09T14:56:00.762825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.751970Z","title":"Executable code actions elicit better LLM agents","venue":null,"work_id":"8be22de1-032c-4fea-a542-8eda5537c085","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.287166Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:24014980529a5a3ce2d82f2e3fd5f078a877d85c71f4df5857c8c8a432893641","observation_id":"df69eb9d-2b8d-4aba-9585-39ff767a9699","resolution":{"observed_at":"2026-08-09T14:56:00.755002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.743499Z","title":"DD-PPO : L earning near-perfect PointGoal navigators from 2.5 billion frames","venue":null,"work_id":"bacb945b-9cc0-4d5d-ab43-b16c6cf230bd","year":2020},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.289910Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:b8adaa026fb699e4fab7e3e44698c65fbffc8d190b2fa5710b0e78d4724c1090","observation_id":"225abf2b-3099-434c-9383-93941f9946cc","resolution":{"observed_at":"2026-08-09T14:56:00.746486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.09009","last_updated":"2025-03-10T23:08:54Z","snapshot_observed_at":"2026-08-14T10:41:41.782496Z","submitted_at":"2024-11-13T20:30:15Z","title":"Cut Your Losses in Large-Vocabulary Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.09009","snapshot_observed_at":"2026-08-09T14:56:00.293146Z","title":"a henb \\","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.293146Z"},"links":{"cited_paper":"/paper/2411.09009","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:8c1299c15be847670b854b300786ff4a16afd655c62ec02446711f6c04e09506","observation_id":"d620ba99-cd16-46b3-8e9a-23932052479a","resolution":{"observed_at":"2026-08-09T14:56:00.293146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.735390Z","title":null,"venue":null,"work_id":"7578d9a2-725c-41d0-a9d7-695752f2a51e","year":1992},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.296441Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:8bfff3991f24960e5006365b45a63bd98c3f3ff94b367e51a50f29fb4356a0ad","observation_id":"21c700fd-8839-4879-b517-dacc14d99b3a","resolution":{"observed_at":"2026-08-09T14:56:00.738369Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-09T14:56:00.299288Z","title":"Qwen2.5 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.299288Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:a38d1ff3b701074a86a18270a04025ba7bb6239a3a7293712ff4d9b9dd7bbef9","observation_id":"7a73299b-08cb-421b-97a5-8b925761da1c","resolution":{"observed_at":"2026-08-09T14:56:00.299288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.726085Z","title":"Intercode: standardizing and benchmarking interactive coding with execution feedback","venue":null,"work_id":"1a744147-73a8-4019-bd5e-d73b4106264e","year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.302540Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:43e94caa59ff1976b00a67b4044221f1deb2821bc0ee4265840158bc68094714","observation_id":"16832fe2-692d-44bf-9b16-297241d0c33e","resolution":{"observed_at":"2026-08-09T14:56:00.729514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.717601Z","title":"Keep CALM and explore: Language models for action generation in text-based games","venue":null,"work_id":"66f65116-f039-480e-b06d-3192dcca4307","year":2020},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.305354Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:7112593dd23e089319fbe2913c64a379897b88cc2736df69926f360ce71108e7","observation_id":"03e338e7-1acd-47d8-a0c4-9cbd0eb20305","resolution":{"observed_at":"2026-08-09T14:56:00.720721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.707742Z","title":"WebShop : Towards scalable real-world web interaction with grounded language agents","venue":null,"work_id":"76e98d2e-fed7-482e-8787-eda34bd16eb3","year":2022},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.308532Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:9a10d7054369eec28b36ffcb5bc7d2e87d652b098361e730962d5d3750dd980f","observation_id":"fbf0f105-5b4d-4ae1-b854-8867a3877da5","resolution":{"observed_at":"2026-08-09T14:56:00.711390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.698293Z","title":"R., and Cao, Y","venue":null,"work_id":"a2a3fe7c-acf4-41a0-8756-8a24d702f3a7","year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.311496Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:92804448b5349975bcafd29be102da9f5706f715d372f51e8c26f709d04489b3","observation_id":"1f641c1d-725b-40a4-9d96-1d1ae04fc5c5","resolution":{"observed_at":"2026-08-09T14:56:00.701404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01825","last_updated":"2023-09-13T03:57:29Z","snapshot_observed_at":"2026-08-16T12:56:40.695367Z","submitted_at":"2023-08-03T15:34:01Z","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01825","snapshot_observed_at":"2026-08-09T14:56:00.314373Z","title":"Scaling relationship on learning mathematical reasoning with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.314373Z"},"links":{"cited_paper":"/paper/2308.01825","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:5676def0d24a0e8bf6f2a84927df426be8870069ae3cae789342278aa053cff8","observation_id":"46657d16-3287-408d-9117-e9057026d4fe","resolution":{"observed_at":"2026-08-09T14:56:00.314373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.689412Z","title":"ST ar: Bootstrapping reasoning with reasoning","venue":null,"work_id":"8dcc3c36-10c3-4819-bdb7-2439c263383e","year":2022},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.317548Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:4a7966a5f355d3fb6776bddbaef4c47be9cd85f51c5d6cf87306162db90ee26f","observation_id":"db68ef43-d623-4056-a443-933467be9fc4","resolution":{"observed_at":"2026-08-09T14:56:00.692776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.680862Z","title":"Fine-tuning large vision-language models as decision-making agents via reinforcement learning","venue":null,"work_id":"292c584a-cda0-424a-9d7d-da16835ce22f","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.320483Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:bd28b428b49cff07febaea04035be66cd083cc34dd345cb1c78a95ea4ce5d8f9","observation_id":"e4826752-2c93-4639-8fb9-f9c367cf720b","resolution":{"observed_at":"2026-08-09T14:56:00.684076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.671130Z","title":"PyTorch FSDP : Experiences on scaling fully sharded data parallel","venue":null,"work_id":"360eb98e-02f4-429d-b1fc-5ad8b25b02dc","year":2023},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.323357Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:87148247f4dd1186118b1f5b31ab166be32f20a7c9e1770b8e4df05406a31389","observation_id":"9745ed1c-470d-4890-bf8a-09c68c3c99f1","resolution":{"observed_at":"2026-08-09T14:56:00.675175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T14:56:00.660694Z","title":"and Zanette, A","venue":null,"work_id":"9e525f99-9f96-40b6-b95a-cb3c5bf287cc","year":2024},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.326131Z"},"links":{"citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:686839c48d4621d05e76f09e5024fceed4d5d6912dbda5ad3251bc3f94b91768","observation_id":"9429a772-2349-4b75-822e-4b617f6fdd67","resolution":{"observed_at":"2026-08-09T14:56:00.664792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08593","last_updated":"2020-01-08T23:02:36Z","snapshot_observed_at":"2026-08-16T00:15:57.597094Z","submitted_at":"2019-09-18T17:33:39Z","title":"Fine-Tuning Language Models from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08593","snapshot_observed_at":"2026-08-09T14:56:00.329027Z","title":"M., Stiennon, N., Wu, J., Brown, T","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-09T14:56:00.329027Z"},"links":{"cited_paper":"/paper/1909.08593","citing_paper":"/paper/2502.01600"},"observation_digest":"sha256:9c9c10b1c2df3165b4da953c631bfc4a7549e1ae32e03d22a496b7af60c6abcc","observation_id":"5208e95e-bee9-4698-a6de-66cfa3ba12d6","resolution":{"observed_at":"2026-08-09T14:56:00.329027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T15:03:39.227865Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":0,"verified_fuzzy":29},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 51 inbound Pith citation observations for arXiv:2502.01600."}