{"as_of":"2026-08-09T13:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:eea4423113eee66ceab3ffdbf00745f4abf692ee22fb24cba6c710ad3e2196da","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:26:55.346259Z","state":"measured"},{"denominator":76,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":76,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:22:18.164811Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T00:55:12.126083Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-07T00:22:18.164811Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.14345","last_updated":"2025-06-17T09:38:45Z","snapshot_observed_at":"2026-08-07T14:48:30.252463Z","submitted_at":"2025-06-17T09:38:45Z","title":"A Vision for Geo-Temporal Deep Research Systems: Towards Comprehensive, Transparent, and Reproducible Geo-Temporal Information Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:22:18.164811Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2506.14345"},"observation_digest":"sha256:b6f557fc4147fc8e570b8be753ee80f8159a0ea7a47d04d4c32900587be97572","observation_id":"38b539da-c53c-43a2-8365-fcfcebcd6817","resolution":{"observed_at":"2026-08-07T00:22:18.164811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-06T10:17:13.579736Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.00271","last_updated":"2025-09-01T02:48:41Z","snapshot_observed_at":"2026-08-09T06:11:59.222373Z","submitted_at":"2025-08-01T02:30:32Z","title":"MetaAgent: Toward Self-Evolving Agent via Tool Meta-Learning","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T10:17:13.579736Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2508.00271"},"observation_digest":"sha256:43430427383276096b56d7f778e8c07d4badc6c2d194971180fe4bb2c05f27a2","observation_id":"d87aa6c1-0808-4f41-9a4d-e7235a8a94d3","resolution":{"observed_at":"2026-08-06T10:17:13.579736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-05T22:46:12.625929Z","title":"Arik, and Jiawei Han","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.06600","last_updated":"2025-08-08T17:55:11Z","snapshot_observed_at":"2026-08-09T01:34:46.047851Z","submitted_at":"2025-08-08T17:55:11Z","title":"BrowseComp-Plus: A More Fair and Transparent Evaluation Benchmark of Deep-Research Agent","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T22:46:12.625929Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2508.06600"},"observation_digest":"sha256:750e98b098698082cdeb86526fd5ab24365efe00dd28cb1c7c58be68a01b3102","observation_id":"11891e05-7f2a-46ad-a05b-74e367e31854","resolution":{"observed_at":"2026-08-05T22:46:12.625929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-05T20:17:09.393341Z","title":"Arik, and Jiawei Han","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.10874","last_updated":"2025-08-14T17:46:01Z","snapshot_observed_at":"2026-08-06T03:19:32.559275Z","submitted_at":"2025-08-14T17:46:01Z","title":"SSRL: Self-Search Reinforcement Learning","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T20:17:09.393341Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2508.10874"},"observation_digest":"sha256:eeb76b01306b50cc54274f0b17d381050441c4dc06a43ac3f0927f9d3520b35d","observation_id":"3e06a672-a89a-4390-a2df-320dc727fe33","resolution":{"observed_at":"2026-08-05T20:17:09.393341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":"2505.15117","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-07-01T00:55:12.126083Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":"7637d87e-ae91-40d6-a208-921791b92408","year":2025},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":244,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:16c99465466c9ea73b0bf006bf82d6b5aa69846b3972e7546524a3be029f591a","observation_id":"9b1aa506-a3b4-4efe-be0a-52109793cfd1","resolution":{"observed_at":"2026-05-18T00:02:24.586905Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":"2505.15117","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-07-01T00:55:12.126083Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":"7637d87e-ae91-40d6-a208-921791b92408","year":2025},"citing_paper":{"arxiv_id":"2510.00861","last_updated":"2026-04-20T08:17:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-01T13:10:36Z","title":"Erase to Improve: Erasable Reinforcement Learning for Search-Augmented LLMs","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-18T11:06:20.058342Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2510.00861"},"observation_digest":"sha256:12f73920ac7090cd48a4c40a35269e3ee5f0f385ab633a51a3c64a5b23423be5","observation_id":"0fcab628-5b5b-4eea-af6c-c1a9c9c9a03d","resolution":{"observed_at":"2026-05-18T11:11:18.283961Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-04T09:50:45.763606Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.13272","last_updated":"2026-06-03T06:16:13Z","snapshot_observed_at":"2026-08-06T23:18:23.213942Z","submitted_at":"2025-10-15T08:17:52Z","title":"Beyond Correctness: Rewarding Faithful Reasoning in Retrieval-Augmented Generation","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-04T09:50:45.763606Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2510.13272"},"observation_digest":"sha256:d65c89b9bbebf692e51f8c73da6fa8959207c48d38eda0c124556c534f36835d","observation_id":"6118e055-e4d6-4a0d-a75a-e3427ebcc2c3","resolution":{"observed_at":"2026-08-04T09:50:45.763606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-04T08:48:40.941733Z","title":"Arik, and Jiawei Han","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.18939","last_updated":"2026-07-14T02:30:28Z","snapshot_observed_at":"2026-08-09T09:17:32.433395Z","submitted_at":"2025-10-21T17:58:45Z","title":"Lost in the Maze: Overcoming Context Limitations in Long-Horizon Agentic Search","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-04T08:48:40.941733Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2510.18939"},"observation_digest":"sha256:8c787b767776aaf4722a0d37219a4d325db2cacb5cc6d89703e3ae8e86dac468","observation_id":"785a30ee-2b24-48a5-af96-a455692b9848","resolution":{"observed_at":"2026-08-04T08:48:40.941733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-03T23:33:45.833809Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.05385","last_updated":"2026-07-23T06:36:18Z","snapshot_observed_at":"2026-08-07T22:51:42.589401Z","submitted_at":"2025-11-07T16:08:34Z","title":"TeaRAG: A Token-Efficient Agentic Retrieval-Augmented Generation Framework","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-03T23:33:45.833809Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2511.05385"},"observation_digest":"sha256:168db9fc1e82ff69bcca3ffcc072949676221806c563638d3fcd84e5a6948a58","observation_id":"85761933-03a2-4669-961f-021236f6c4fc","resolution":{"observed_at":"2026-08-03T23:33:45.833809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-03T17:53:54.804421Z","title":"O., and Han, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.07795","last_updated":"2026-05-30T17:04:27Z","snapshot_observed_at":"2026-08-08T10:56:08.402825Z","submitted_at":"2025-12-08T18:26:58Z","title":"ReasonBENCH: Benchmarking the (In)Stability of LLM Reasoning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T17:53:54.804421Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2512.07795"},"observation_digest":"sha256:b12587be1e86a69302a21195abebea6fceb04ffb5ad5f55fd61f655e50b7b63e","observation_id":"f189d6d9-658d-4f1c-98a6-3b1e506dc583","resolution":{"observed_at":"2026-08-03T17:53:54.804421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":"2505.15117","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-07-01T00:55:12.126083Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":"7637d87e-ae91-40d6-a208-921791b92408","year":2025},"citing_paper":{"arxiv_id":"2605.01248","last_updated":"2026-06-08T23:39:22Z","snapshot_observed_at":"2026-07-06T23:14:29.125042Z","submitted_at":"2026-05-02T05:01:05Z","title":"$S^3$-R1: Learning to Retrieve and Answer Step-by-Step with Synthetic Data","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-09T15:08:53.731480Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2605.01248"},"observation_digest":"sha256:9af8914f45642263690cb9ccb41995a7508d02fbaf135e3b876e2519c7419da3","observation_id":"7cb12523-4345-4d84-a68f-d2958bf61d59","resolution":{"observed_at":"2026-05-11T16:46:06.919257Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":"2505.15117","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-07-01T00:55:12.126083Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":"7637d87e-ae91-40d6-a208-921791b92408","year":2025},"citing_paper":{"arxiv_id":"2605.01248","last_updated":"2026-06-08T23:39:22Z","snapshot_observed_at":"2026-07-06T23:14:29.125042Z","submitted_at":"2026-05-02T05:01:05Z","title":"$S^3$-R1: Learning to Retrieve and Answer Step-by-Step with Synthetic Data","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-01T00:48:54.797750Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2605.01248"},"observation_digest":"sha256:d652b1f8c120a8bc0b148ff2d3a69f2609f6581f6c90e5ff4261c249d3e80337","observation_id":"600bc524-c3a1-494f-99ec-69a625a52d40","resolution":{"observed_at":"2026-07-01T00:55:12.127996Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":"2505.15117","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-07-01T00:55:12.126083Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":"7637d87e-ae91-40d6-a208-921791b92408","year":2025},"citing_paper":{"arxiv_id":"2605.01399","last_updated":"2026-05-02T11:43:23Z","snapshot_observed_at":"2026-07-06T23:14:38.379875Z","submitted_at":"2026-05-02T11:43:23Z","title":"Verbal-R3: Verbal Reranker as the Missing Bridge between Retrieval and Reasoning","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-09T14:40:23.170932Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2605.01399"},"observation_digest":"sha256:4096547f93d2d08701d268fc14184a8fd2ad47ed57811d6d5193077a4cf92ccd","observation_id":"d8193e3d-2d96-468a-b0f0-524b0649de2b","resolution":{"observed_at":"2026-05-11T16:51:08.796266Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":"2505.15117","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-07-01T00:55:12.126083Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":"7637d87e-ae91-40d6-a208-921791b92408","year":2025},"citing_paper":{"arxiv_id":"2605.06285","last_updated":"2026-05-07T13:56:13Z","snapshot_observed_at":"2026-07-06T23:18:46.145104Z","submitted_at":"2026-05-07T13:56:13Z","title":"LatentRAG: Latent Reasoning and Retrieval for Efficient Agentic RAG","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-08T10:27:00.257353Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2605.06285"},"observation_digest":"sha256:a08b36876c2adb01950c756d277fe6ff12501ef28edd9bc4d303e01a3a8c66eb","observation_id":"e03c2656-8aee-4852-8bbf-20f7a3d434a2","resolution":{"observed_at":"2026-05-11T20:01:12.045981Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":"2505.15117","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-07-01T00:55:12.126083Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents","venue":null,"work_id":"7637d87e-ae91-40d6-a208-921791b92408","year":2025},"citing_paper":{"arxiv_id":"2606.26122","last_updated":"2026-05-27T21:21:42Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-05-27T21:21:42Z","title":"DocArena: Turning Raw Documents into Controllable Training Environments for Document Search Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T12:50:16.625077Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2606.26122"},"observation_digest":"sha256:c0ecab9c32cc3b98040b5d2ebac4fda16f9e47af49381e0ce4609d964492bd43","observation_id":"8fcae104-3086-451d-888d-490e03e36875","resolution":{"observed_at":"2026-06-29T12:53:26.519864Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-02T03:10:51.532947Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13988","last_updated":"2026-07-15T16:16:42Z","snapshot_observed_at":"2026-08-08T08:46:44.738945Z","submitted_at":"2026-07-15T16:16:42Z","title":"TRACE: Turn-level Reward Assignment via Credit Estimation for Long-Horizon Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T03:10:51.532947Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2607.13988"},"observation_digest":"sha256:07b45a267781edb20d427a02889d441aeafd004334bc9e1840d678b9f55a537f","observation_id":"fe7c2143-8f14-4272-87fe-6dcc857aabba","resolution":{"observed_at":"2026-08-02T03:10:51.532947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15117","snapshot_observed_at":"2026-08-04T23:23:14.639942Z","title":"An empirical study on reinforcement learning for reasoning-search interleaved llm agents.arXiv preprint arXiv:2505.15117, 2025a","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.01667","last_updated":"2026-08-03T04:01:36Z","snapshot_observed_at":"2026-08-07T12:28:19.531850Z","submitted_at":"2026-08-03T04:01:36Z","title":"TCPO: Turn-Level Credit Policy Optimization","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-04T23:23:14.639942Z"},"links":{"cited_paper":"/paper/2505.15117","citing_paper":"/paper/2608.01667"},"observation_digest":"sha256:0750067fedc1515a344fa56275c590b4fd70ebeb9637d70311b7d26d650a5f46","observation_id":"3f450e9f-b910-4ae1-be33-eae85bac52ea","resolution":{"observed_at":"2026-08-04T23:23:14.639942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.15117/citation-record","integrity":"/paper/2505.15117/integrity","json":"/paper/2505.15117/citation-record.json","paper":"/paper/2505.15117"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T15:26:49.226134Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.226134Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:6c7f06a5ce02f643753cc6a0a08c7c772aa2a89a1e60a7f8fe8b08dd717da275","observation_id":"a9c57f54-6a3f-411f-9cd5-c481acc54943","resolution":{"observed_at":"2026-08-07T15:26:49.226134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-07-06T17:34:07.737296Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-08-07T15:26:49.330797Z","title":"Back to basics: Revisiting reinforce style optimization for learning from human feedback in llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.330797Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:fb1145afdcdb906043dbc49142fde4c9415b223c30a36d0cbc6edc1a687aacd6","observation_id":"4b28661e-4667-436f-ad52-9e167fd09ce4","resolution":{"observed_at":"2026-08-07T15:26:49.330797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.078920Z","title":"Self-rag: Learn- ing to retrieve, generate, and critique through self-reflection","venue":null,"work_id":"aaa58aed-a81e-4945-a705-46f41b1059fe","year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.410784Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:d89d2251eb7e662c060370fbbdd764317477d113a246583ea7f0ea7c5392a1b9","observation_id":"1fbacb48-e9af-47ef-884b-eb32fcb592d5","resolution":{"observed_at":"2026-08-07T15:26:59.157980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19470","last_updated":"2025-09-23T03:45:42Z","snapshot_observed_at":"2026-08-09T02:50:54.082128Z","submitted_at":"2025-03-25T09:00:58Z","title":"ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19470","snapshot_observed_at":"2026-08-07T15:26:49.507040Z","title":"Research: Learning to reason with search for llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.507040Z"},"links":{"cited_paper":"/paper/2503.19470","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:d64217898ffee36d209ef236ba13f05f1ef54f35f2484846362120ae13ab0736","observation_id":"3c45f28c-547c-4ea1-a66b-c9457ded3b34","resolution":{"observed_at":"2026-08-07T15:26:49.507040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1907.09190","last_updated":"2019-07-22T09:01:35Z","snapshot_observed_at":"2026-08-03T09:05:08.518348Z","submitted_at":"2019-07-22T09:01:35Z","title":"ELI5: Long Form Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.09190","snapshot_observed_at":"2026-08-07T15:26:49.595880Z","title":"Eli5: Long form question answering","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.595880Z"},"links":{"cited_paper":"/paper/1907.09190","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:4152ca0da0732f551472dd9272c8bbf58d7921e072fd896b8a914238e86dee48","observation_id":"4f6f5d45-770b-499d-a63b-ccc844135ad7","resolution":{"observed_at":"2026-08-07T15:26:49.595880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01307","last_updated":"2025-08-15T15:21:46Z","snapshot_observed_at":"2026-08-08T21:52:29.510852Z","submitted_at":"2025-03-03T08:46:22Z","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01307","snapshot_observed_at":"2026-08-07T15:26:49.679749Z","title":"Cognitive behaviors that enable self-improving reasoners, or, four habits of highly effective stars","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.679749Z"},"links":{"cited_paper":"/paper/2503.01307","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:9b1eb0a3b90b20dd5b74d99ef10001018b50fec8822154a2527150b950975249","observation_id":"7b898ac7-96ce-4443-a0e5-0a274b00fb5d","resolution":{"observed_at":"2026-08-07T15:26:49.679749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10997","last_updated":"2024-03-27T09:16:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-18T07:47:33Z","title":"Retrieval-Augmented Generation for Large Language Models: A Survey","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10997","snapshot_observed_at":"2026-08-07T15:26:49.769546Z","title":"Retrieval-augmented generation for large language models: A survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.769546Z"},"links":{"cited_paper":"/paper/2312.10997","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:ec740fc7d49573f1afd6935f5092c0198a88bf266ed140908b7e40aa7fa62304","observation_id":"56345dbc-a3ed-4a1c-bf4e-68a59f2c5470","resolution":{"observed_at":"2026-08-07T15:26:49.769546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.08433","last_updated":"2023-10-12T15:56:24Z","snapshot_observed_at":"2026-07-06T16:32:00.957275Z","submitted_at":"2023-10-12T15:56:24Z","title":"A Confederacy of Models: a Comprehensive Evaluation of LLMs on Creative Writing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.08433","snapshot_observed_at":"2026-08-07T15:26:49.877999Z","title":"A confederacy of models: A comprehensive evaluation of llms on creative writing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.877999Z"},"links":{"cited_paper":"/paper/2310.08433","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:036edf47ed4548f0147f6c1b7fc5c4e003aa4d6792a5ef927e29cbd7eb183639","observation_id":"a9547765-daaf-4d6a-b000-24d7e114b491","resolution":{"observed_at":"2026-08-07T15:26:49.877999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T15:26:49.967806Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:49.967806Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:d2a3daea93e711bf55b122697d5acf25cc2852d95b7a7f42852f6b45280c60c4","observation_id":"585497cc-a9c9-4f4b-81bb-fa58ff551145","resolution":{"observed_at":"2026-08-07T15:26:49.967806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.15556","last_updated":"2022-03-29T13:38:03Z","snapshot_observed_at":"2026-07-06T12:54:11.616335Z","submitted_at":"2022-03-29T13:38:03Z","title":"Training Compute-Optimal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.15556","snapshot_observed_at":"2026-08-07T15:26:50.082149Z","title":"Training compute-optimal large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.082149Z"},"links":{"cited_paper":"/paper/2203.15556","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:27b5acee74daf18dbaab15d254534320187d1f076cb3aa181f368a610679db69","observation_id":"b7630dd5-73ac-4352-be50-f252bdc34117","resolution":{"observed_at":"2026-08-07T15:26:50.082149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.912426Z","title":"Long-context llms meet rag: Overcoming challenges for long inputs in rag","venue":null,"work_id":"2b14a3cb-e1a6-413c-8382-7e04a5cf940c","year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.196189Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:505536c7751b834e11de136edb1be512cdffd537e679ca156cba5f1e677a82e9","observation_id":"f6534b44-fd37-4592-977e-7376e1c91ff2","resolution":{"observed_at":"2026-08-07T15:26:58.985496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03699","last_updated":"2025-07-23T21:26:26Z","snapshot_observed_at":"2026-08-09T04:26:07.182723Z","submitted_at":"2025-02-06T01:22:06Z","title":"LLM Alignment as Retriever Optimization: An Information Retrieval Perspective","version":3},"cited_work":{"arxiv_id":"2502.03699","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.03699","snapshot_observed_at":"2026-08-07T15:26:55.832369Z","title":"LLM Alignment as Retriever Optimization: An Information Retrieval Perspective","venue":"cs.CL","work_id":"48287252-8e15-42ae-87b0-e1262610ac90","year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.289327Z"},"links":{"cited_paper":"/paper/2502.03699","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:75ebec779ef5bc806010f0ba2a0dbaa635ba0a6a616d7bf05b076cffab9117ed","observation_id":"717a137d-4795-4d92-aa45-2f3bd6b12ce5","resolution":{"observed_at":"2026-08-07T15:26:55.887278Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09516","last_updated":"2025-08-05T19:08:38Z","snapshot_observed_at":"2026-07-06T20:51:28.022519Z","submitted_at":"2025-03-12T16:26:39Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09516","snapshot_observed_at":"2026-08-07T15:26:50.403376Z","title":"Search-r1: Training llms to reason and leverage search engines with reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.403376Z"},"links":{"cited_paper":"/paper/2503.09516","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:a4aa60008488db1049c466cf4a1b99ef02c54984042580983292655472dfaea0","observation_id":"b058ef94-0010-4640-97d2-ae7ecdca5340","resolution":{"observed_at":"2026-08-07T15:26:50.403376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.770379Z","title":"Reinforcement learning: A survey","venue":null,"work_id":"ea74ade4-7d8b-4fb8-82c9-1a4e620ec0c8","year":1996},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.494200Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:62421b60c23a9f8711c94cc52e967286467dd557ed5caeff494147969ba38e43","observation_id":"c3821500-bdf1-4fca-a0d4-f43e0ba4cc4c","resolution":{"observed_at":"2026-08-07T15:26:58.841646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-07T15:26:50.590956Z","title":"Scaling laws for neural language models","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.590956Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:e9f50c48d7c956bf3f5e869dfc2b8569c6d92d679370a3224b15848298f17a50","observation_id":"19e1688c-cfa7-418e-a7fc-348c0c3c9b8c","resolution":{"observed_at":"2026-08-07T15:26:50.590956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.650590Z","title":"Dense passage retrieval for open-domain question answering","venue":null,"work_id":"86fe5f5a-502d-40de-9df9-14c272276c8d","year":2020},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.679252Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:3a60a4af516b48985a24684db71745ddcffeda34b5333616bf0754d7d9ff96c6","observation_id":"03d37d64-fccf-4d4d-aba6-85a748414895","resolution":{"observed_at":"2026-08-07T15:26:58.703594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:50.748824Z","title":"A survey of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.748824Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:82d200d7dd61c70ff903498ff4fb90a829907c05b98f80b0ac4cc44c50164c08","observation_id":"ca007945-7e15-4249-a24c-414673a7b03d","resolution":{"observed_at":"2026-08-07T15:26:50.748824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:50.837507Z","title":"Natural questions: a benchmark for question answering research","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.837507Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:8faa5d38520b5accb7ae1d45f461adb7977655731c8d6b8d0f20579858440a62","observation_id":"4c94b57a-382e-430c-a0c6-0d0c5fc7a1a3","resolution":{"observed_at":"2026-08-07T15:26:50.837507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:50.909895Z","title":"Med-r1: Reinforce- ment learning for generalizable medical reasoning in vision-language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.909895Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:d5e28d15df0e4d79ac64813300895ba35624e8ec3326496dfb143e44181a91b0","observation_id":"17f0a0b5-8016-4d30-81f7-4fe71c83a2e4","resolution":{"observed_at":"2026-08-07T15:26:50.909895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13787","last_updated":"2024-06-08T16:40:12Z","snapshot_observed_at":"2026-08-02T18:11:57.036767Z","submitted_at":"2024-03-20T17:49:54Z","title":"RewardBench: Evaluating Reward Models for Language Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13787","snapshot_observed_at":"2026-08-07T15:26:50.989095Z","title":"Rewardbench: Evaluating reward models for language modeling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:50.989095Z"},"links":{"cited_paper":"/paper/2403.13787","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:fabadb0d7a4b2987918440a2b36e23cd214abf48cce757ee76cf03c9d5e7088c","observation_id":"1f416758-8579-4885-887e-dad10bb3f9d6","resolution":{"observed_at":"2026-08-07T15:26:50.989095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.519296Z","title":"Large language models in finance: A survey","venue":null,"work_id":"5483cc8f-b359-446d-88ac-4e2ea5103e5a","year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.109969Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:c689b2d463fb4e88cdea538919956c4833332b502179f3fbd9d6e8164c5b5e5f","observation_id":"2f145531-f0a2-41ec-880f-42eba51ab072","resolution":{"observed_at":"2026-08-07T15:26:58.568392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:51.227945Z","title":"Rec-r1: Bridging generative large language mod- els and user-centric recommendation systems via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.227945Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:867856e4932ef4d360cde78d2a3f4eaf3e3f306786998d2ae9d4470186806f0b","observation_id":"c74cf3b1-65ef-40d7-b028-718a802ccbb9","resolution":{"observed_at":"2026-08-07T15:26:51.227945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.399385Z","title":"Ra-dit: Retrieval-augmented dual instruction tuning","venue":null,"work_id":"d6bad3e6-5500-42f3-9359-b37525327544","year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.320200Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:c8d18d21fe1bd76fecb1fb01e77dc8eee8d80634d8f14db0317abed691a4d588","observation_id":"31f92054-4f7e-4285-b9cf-38cd12aa6c34","resolution":{"observed_at":"2026-08-07T15:26:58.456715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14239","last_updated":"2025-04-19T09:25:55Z","snapshot_observed_at":"2026-08-05T23:52:52.919433Z","submitted_at":"2025-04-19T09:25:55Z","title":"InfiGUI-R1: Advancing Multimodal GUI Agents from Reactive Actors to Deliberative Reasoners","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14239","snapshot_observed_at":"2026-08-07T15:26:51.429733Z","title":"Infigui-r1: Advancing multimodal gui agents from reactive actors to deliberative reasoners","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.429733Z"},"links":{"cited_paper":"/paper/2504.14239","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:e19032109a37b89e46b4e1b76d7b14aa1caa40770eff1e51289347c8a23ead15","observation_id":"fca8fc0a-f564-42ca-9bd9-479b52f91105","resolution":{"observed_at":"2026-08-07T15:26:51.429733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:51.566362Z","title":"Fin-r1: A large language model for financial reasoning through reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.566362Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:ac517a7899de25b9d317fb8974457ecc4db182e8a950c3e08dae55f5b66ee9a4","observation_id":"341d1ffc-de10-4f4f-8fa8-20f9a82917f2","resolution":{"observed_at":"2026-08-07T15:26:51.566362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.258767Z","title":"Chatqa: Surpassing gpt-4 on conversational qa and rag","venue":null,"work_id":"53c1afe5-d421-4ba8-a2e9-d05b20993f0c","year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.658947Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:e703c160693b7acde681c0f39e01308ab333ae8f8d175365635b5efc6955208d","observation_id":"5f61c753-65e0-45a0-ab56-5b0dee9f94a6","resolution":{"observed_at":"2026-08-07T15:26:58.324495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.116093Z","title":"Efficient and robust approximate nearest neighbor search using hierarchical navigable small world graphs","venue":null,"work_id":"1b460e64-d434-43bf-9e3e-c0b13de68987","year":2018},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.746406Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:bdc0fb7a51d2810bcc282a8c6cc41e7b7ef1ba39d6c4ed6daf1595a98d3635ea","observation_id":"e7884401-3cf6-45d4-ad39-d78d65af419e","resolution":{"observed_at":"2026-08-07T15:26:58.190378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.908461Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"d5805ab5-030f-46ce-a199-a9fa4d9e709a","year":2022},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.844156Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:38ac1dc87350d0c070145139b5d772aca452066bcb420c7061b6214e7c910e79","observation_id":"595bedd6-b420-4b9a-827b-64e5da074ae9","resolution":{"observed_at":"2026-08-07T15:26:57.990089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.761544Z","title":"A study of generative large language model for medical research and healthcare","venue":null,"work_id":"17d974f9-46b7-487b-8838-1875d784e990","year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:51.933207Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:58cae18afacc9c512180a5f9c88a894d7e4a130b84de5e7c0aecb3636a98d6ff","observation_id":"195f0ec1-8ef3-4d27-9893-765da6383166","resolution":{"observed_at":"2026-08-07T15:26:57.826853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03350","last_updated":"2023-10-17T18:57:17Z","snapshot_observed_at":"2026-07-31T15:52:06.089691Z","submitted_at":"2022-10-07T06:50:23Z","title":"Measuring and Narrowing the Compositionality Gap in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03350","snapshot_observed_at":"2026-08-07T15:26:52.033954Z","title":"Measuring and narrowing the compositionality gap in language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.033954Z"},"links":{"cited_paper":"/paper/2210.03350","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:01257b1b5ebd8a7c477d7aaaeaa5e64c89a5ef7aee9b7106d7269399f39550d1","observation_id":"65ef205c-a837-428d-8b96-ef6feb42b081","resolution":{"observed_at":"2026-08-07T15:26:52.033954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:52.134087Z","title":"Direct preference optimization: Your language model is secretly a reward model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.134087Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:49a6e798795cc1c9efd5829fc35e4dde94604bf333891059d44d3c8980677a39","observation_id":"5d464510-268c-4b36-b755-855e0b957174","resolution":{"observed_at":"2026-08-07T15:26:52.134087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.566787Z","title":"Gpqa: A graduate-level google-proof q&a benchmark","venue":null,"work_id":"c4403cc5-9f6f-4164-ae06-d838f52bd310","year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.283836Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:64aa4b11d4c8fb7d07dea9bf29c2600638b8b57af2fe2a1b6371442641aae2d1","observation_id":"06269762-9b6c-417f-aa33-40161e185fee","resolution":{"observed_at":"2026-08-07T15:26:57.661037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.378248Z","title":"The probabilistic relevance framework: Bm25 and beyond","venue":null,"work_id":"12d025b7-c5e6-4a00-9e45-bfc543a222ca","year":2009},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.398976Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:4a0799d37f4d6c14aeb1ba9278205465455cc8551d13425be76827caaed18410","observation_id":"900386d5-fcba-4598-9226-f38f09ab511a","resolution":{"observed_at":"2026-08-07T15:26:57.466025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.169448Z","title":"Toolformer: Language models can teach themselves to use tools","venue":null,"work_id":"0b584c54-c88d-414b-b672-da78be033a11","year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.535004Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:78c88307d8a4036e9c681fd6c88d7b7e6767680c4424d3c859d96e96856dcd74","observation_id":"40aa8960-f10e-434e-bb78-93857c0be9d5","resolution":{"observed_at":"2026-08-07T15:26:57.242610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T15:26:52.616549Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.616549Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:edeada64bb44db2a6e7704186bde058ffc46af7fa0e62958931de67bc7156079","observation_id":"6b9a6faf-50e9-4218-a32b-4af6ddb29e7d","resolution":{"observed_at":"2026-08-07T15:26:52.616549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09136","last_updated":"2026-04-01T15:51:06Z","snapshot_observed_at":"2026-08-07T12:36:27.102579Z","submitted_at":"2025-01-15T20:40:25Z","title":"Agentic Retrieval-Augmented Generation: A Survey on Agentic RAG","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09136","snapshot_observed_at":"2026-08-07T15:26:52.754080Z","title":"Agentic retrieval-augmented generation: A survey on agentic rag","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.754080Z"},"links":{"cited_paper":"/paper/2501.09136","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:91afee0c8b83e15b0bfd34fb684fac2fb072723c93881f6069ca92f33de8ee2d","observation_id":"9023e9e1-0bed-4620-812b-fabe429d2868","resolution":{"observed_at":"2026-08-07T15:26:52.754080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.05592","last_updated":"2025-03-18T08:32:24Z","snapshot_observed_at":"2026-08-09T05:07:17.305244Z","submitted_at":"2025-03-07T17:14:44Z","title":"R1-Searcher: Incentivizing the Search Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.05592","snapshot_observed_at":"2026-08-07T15:26:52.868730Z","title":"R1-searcher: Incentivizing the search capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.868730Z"},"links":{"cited_paper":"/paper/2503.05592","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:7ff8f6b8c1678dfb170fc57b9aec16fc1debfcc537caf1e5b124cdf8e266688a","observation_id":"b6a6294c-588d-4f46-8ec0-7470b7666aad","resolution":{"observed_at":"2026-08-07T15:26:52.868730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.06092","last_updated":"2023-01-22T14:25:40Z","snapshot_observed_at":"2026-07-06T12:59:46.149474Z","submitted_at":"2022-04-12T21:58:44Z","title":"ASQA: Factoid Questions Meet Long-Form Answers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.06092","snapshot_observed_at":"2026-08-07T15:26:52.983403Z","title":"Asqa: Factoid questions meet long-form answers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:52.983403Z"},"links":{"cited_paper":"/paper/2204.06092","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:719845107d4de57cf082eab4e80148e69665a4543f888b3fa874c4c47e72f12c","observation_id":"7c40fc6e-ebcd-4414-b549-c8fb80e980cf","resolution":{"observed_at":"2026-08-07T15:26:52.983403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.962330Z","title":"Reinforcement learning","venue":null,"work_id":"e8146060-b731-40ec-b84a-2aded08d2eef","year":1999},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.080773Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:f0aeb876f7105c52f0316f0d1b1105d4b008a2bbb2d5daa409600cc3e93cb71f","observation_id":"3cd1d6d0-871e-4444-896f-883123a0a9ee","resolution":{"observed_at":"2026-08-07T15:26:57.086436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T15:26:53.199033Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.199033Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:d80cea5f233ec05e56775348f670df801c358e20ed2aca9130abc2fcf9ea7517","observation_id":"4333ad5d-3a13-4dc5-b02c-a338220a6faf","resolution":{"observed_at":"2026-08-07T15:26:53.199033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10509","last_updated":"2023-06-23T00:59:13Z","snapshot_observed_at":"2026-07-06T14:33:08.041820Z","submitted_at":"2022-12-20T18:26:34Z","title":"Interleaving Retrieval with Chain-of-Thought Reasoning for Knowledge-Intensive Multi-Step Questions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10509","snapshot_observed_at":"2026-08-07T15:26:53.311795Z","title":"Interleaving retrieval with chain-of-thought reasoning for knowledge-intensive multi-step questions","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.311795Z"},"links":{"cited_paper":"/paper/2212.10509","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:61a5196d07cec567ceca543ee4738e2824c04a49601bd5e4d6edf3e7f8cbb120","observation_id":"9903980d-f83a-470c-b809-9297b5fa6ff9","resolution":{"observed_at":"2026-08-07T15:26:53.311795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.03533","last_updated":"2024-02-22T06:21:51Z","snapshot_observed_at":"2026-07-06T14:27:46.217000Z","submitted_at":"2022-12-07T09:25:54Z","title":"Text Embeddings by Weakly-Supervised Contrastive Pre-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.03533","snapshot_observed_at":"2026-08-07T15:26:53.451848Z","title":"Text embeddings by weakly-supervised contrastive pre-training","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.451848Z"},"links":{"cited_paper":"/paper/2212.03533","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:a77867b61996884204acce105cd4d03618e92a40baccdc8f63d6b647686f1235","observation_id":"f818db25-cce8-4a26-b6db-dc0742739ac5","resolution":{"observed_at":"2026-08-07T15:26:53.451848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20571","snapshot_observed_at":"2026-08-07T15:26:53.529810Z","title":"Reinforcement learning for reasoning in large language models with one training example","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.529810Z"},"links":{"cited_paper":"/paper/2504.20571","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:92b8f6b5ca1a4513b812a076279348a8f9c4d63e8f8ad575907357d36c813ab3","observation_id":"13cd128c-5533-4061-9103-9dd198580236","resolution":{"observed_at":"2026-08-07T15:26:53.529810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-08-08T11:44:11.313729Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-07T15:26:53.665506Z","title":"Ragen: Understanding self-evolution in llm agents via multi-turn reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.665506Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:ebf7bc43f3ab99237aa292c0567554c53a07de319fdac736d3c88e0941496f82","observation_id":"1454a45e-770d-4911-bd2a-429324bdc637","resolution":{"observed_at":"2026-08-07T15:26:53.665506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.758476Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":"6ca36448-29e7-4683-8424-48c95789693b","year":2022},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.780369Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:dbb239e11aa8329d7658ef4e39ff84603bfd68b5add2c8fbe7359ca4e159cefe","observation_id":"e68b8865-606f-4ebd-82f2-24226fc8b8a3","resolution":{"observed_at":"2026-08-07T15:26:56.850656Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04368","last_updated":"2024-11-07T01:58:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T01:58:42Z","title":"Measuring short-form factuality in large language models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04368","snapshot_observed_at":"2026-08-07T15:26:53.898164Z","title":"Measuring short-form factuality in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:53.898164Z"},"links":{"cited_paper":"/paper/2411.04368","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:cc0e247df899da363bf1f8e5d438d062538d3769469d12d0d0e53ea98ecbf742","observation_id":"29f35dad-8602-4e0c-91fc-4b05780a2a6c","resolution":{"observed_at":"2026-08-07T15:26:53.898164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.595090Z","title":"Simple statistical gradient-following algorithms for connectionist reinforce- ment learning","venue":null,"work_id":"a9368cb4-7048-4427-80fb-1b76f122067b","year":1992},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.004145Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:3633cf166883fec1ff95c77e19157ba463ed21e21dfdd4ee90d2bde4f463c8ca","observation_id":"f124a3f5-834f-435a-9c0e-b74acc1ed4fd","resolution":{"observed_at":"2026-08-07T15:26:56.675204Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10458","last_updated":"2025-10-01T04:55:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:45:54Z","title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10458","snapshot_observed_at":"2026-08-07T15:26:54.106018Z","title":"Gui-r1: A generalist r1-style vision-language action model for gui agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.106018Z"},"links":{"cited_paper":"/paper/2504.10458","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:c23c966692e6fa21c68c40f0ab1f273db4289601c8f04779c10c7bb58354dffa","observation_id":"bfd1f691-7fa0-4e1b-8cd1-5f61af21abd6","resolution":{"observed_at":"2026-08-07T15:26:54.106018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14768","last_updated":"2025-02-20T17:49:26Z","snapshot_observed_at":"2026-08-04T12:32:53.090165Z","submitted_at":"2025-02-20T17:49:26Z","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14768","snapshot_observed_at":"2026-08-07T15:26:54.279168Z","title":"Logic-rl: Unleashing llm reasoning with rule-based reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.279168Z"},"links":{"cited_paper":"/paper/2502.14768","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:11861a6844171f24992e1510260f06e23700fb580244db7d044cf882820265ec","observation_id":"0be230af-cc11-4b04-9d36-a1791b7f2d74","resolution":{"observed_at":"2026-08-07T15:26:54.279168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T15:26:54.391914Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.391914Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:052fd02b089cb600617ebd77411c7384b6c6e419ba5197f9226cda5d94ef7c58","observation_id":"92849fa8-cbaf-4c54-a339-5130b46e8ab6","resolution":{"observed_at":"2026-08-07T15:26:54.391914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.09600","last_updated":"2018-09-25T17:28:20Z","snapshot_observed_at":"2026-08-09T10:11:58.541007Z","submitted_at":"2018-09-25T17:28:20Z","title":"HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.09600","snapshot_observed_at":"2026-08-07T15:26:54.470888Z","title":"Hotpotqa: A dataset for diverse, explainable multi-hop question answering","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.470888Z"},"links":{"cited_paper":"/paper/1809.09600","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:68b7a6056622447fe988e5bc85c6b72402a8c0c4d8c96f6d1a8fcd054203c2a5","observation_id":"0e2d6c4b-4939-4028-9f89-0c0439cf2f94","resolution":{"observed_at":"2026-08-07T15:26:54.470888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.403140Z","title":"React: Synergizing reasoning and acting in language models","venue":null,"work_id":"792fce9e-3bd2-46b0-955f-d5097e669651","year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.608235Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:1dbac33ac200ede5f8a8e745fd098d77f9b2ec9408992368356346f83e76521a","observation_id":"de6b9cce-e7d1-4489-ae11-6655f4e0edc3","resolution":{"observed_at":"2026-08-07T15:26:56.510314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-07T15:26:54.687871Z","title":"Dapo: An open-source llm reinforcement learning system at scale","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.687871Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:1d1a10c517c4712d64119199e0f72b822de7b97a1ffba12e2dafea29cdd7f712","observation_id":"33aab401-0b34-4dc2-9521-4f7dd023735e","resolution":{"observed_at":"2026-08-07T15:26:54.687871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05118","last_updated":"2025-04-11T02:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T14:21:11Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.05118","snapshot_observed_at":"2026-08-07T15:26:54.812847Z","title":"Vapo: Efficient and reliable reinforcement learning for advanced reasoning tasks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.812847Z"},"links":{"cited_paper":"/paper/2504.05118","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:aa31ec012d1f2d70505de9c1adad4f5d65e81620c0d3fac45f44400afff6342e","observation_id":"aa196788-e30a-4150-803f-794700ddbea6","resolution":{"observed_at":"2026-08-07T15:26:54.812847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18013","last_updated":"2025-03-23T10:21:14Z","snapshot_observed_at":"2026-08-07T16:43:26.268417Z","submitted_at":"2025-03-23T10:21:14Z","title":"Vision-R1: Evolving Human-Free Alignment in Large Vision-Language Models via Vision-Guided Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18013","snapshot_observed_at":"2026-08-07T15:26:54.929294Z","title":"Vision-r1: Evolving human-free alignment in large vision-language models via vision- guided reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:54.929294Z"},"links":{"cited_paper":"/paper/2503.18013","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:46d9d7fff9c5dc98134f72ab480aa5045a146e5084f0f253a91cf253e31c6db6","observation_id":"05a4234d-b062-4944-b633-d5c847f3be3c","resolution":{"observed_at":"2026-08-07T15:26:54.929294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.182541Z","title":"Benchmarking large language models for news summarization","venue":null,"work_id":"cfe329dc-0ef3-498c-b981-304eadb7ad85","year":2024},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:55.004398Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:5bcf68d340eb96733f51709d5268dbad35db829b0b8a63c9b73b26481563f11c","observation_id":"e3747c8b-8768-4e14-af6c-d6ce749ab8c3","resolution":{"observed_at":"2026-08-07T15:26:56.319879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-07T15:26:55.141146Z","title":"A survey of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:55.141146Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:e28763e7f4569c1b51e7d0d8a0fbdc4e97ccfdfc4504c26ac9d71fbdf598f506","observation_id":"a765309a-41f1-4b5f-a97f-40bee11c104a","resolution":{"observed_at":"2026-08-07T15:26:55.141146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.03160","last_updated":"2025-04-17T04:46:08Z","snapshot_observed_at":"2026-07-06T21:04:06.413573Z","submitted_at":"2025-04-04T04:41:28Z","title":"DeepResearcher: Scaling Deep Research via Reinforcement Learning in Real-world Environments","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.03160","snapshot_observed_at":"2026-08-07T15:26:55.223022Z","title":"<| im_start | > as si sta nt","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:55.223022Z"},"links":{"cited_paper":"/paper/2504.03160","citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:9c17b8eb19c8f7c0243ec2cda4ed3ad1e81a1d19ac8d8c8f4c221ff9d074011f","observation_id":"46073e24-a2eb-43b6-9e9a-90fc8e3369fd","resolution":{"observed_at":"2026-08-07T15:26:55.223022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.025148Z","title":"Delicatessen","venue":null,"work_id":"b1986ac8-1659-47c4-aa8e-3c2d8c89e5d1","year":1991},"citing_paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents","version":1},"reference_index":1953,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:55.346259Z"},"links":{"citing_paper":"/paper/2505.15117"},"observation_digest":"sha256:d94f011b549107ce12cacabd457d47637a470f306b9f7cf86e7dd3a158c70d62","observation_id":"6bb5f631-5897-445b-b8eb-09fd812f0456","resolution":{"observed_at":"2026-08-07T15:26:56.105243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.15117","last_updated":"2025-05-21T05:09:43Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T04:26:25.608959Z","submitted_at":"2025-05-21T05:09:43Z","title":"An Empirical Study on Reinforcement Learning for Reasoning-Search Interleaved LLM Agents"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":1,"verified_fuzzy":19},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 17 inbound Pith citation observations for arXiv:2505.15117."}