{"as_of":"2026-08-21T04:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0dedbe94df1e6255439aed7b5d484949e0ed72d79501dbd39e86cb8462103339","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":91,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":91,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":91,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":91,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:16:31.845707Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":3,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-12T18:28:49.296695Z","title":"P., Kawaguchi, K., and Shieh, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.11504","last_updated":"2024-11-18T12:04:52Z","snapshot_observed_at":"2026-08-19T02:03:59.847880Z","submitted_at":"2024-11-18T12:04:52Z","title":"Search, Verify and Feedback: Towards Next Generation Post-training Paradigm of Foundation Models via Verifier Engineering","version":1},"reference_index":135,"source":"arxiv_source","source_observed_at":"2026-08-12T18:28:49.296695Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2411.11504"},"observation_digest":"sha256:9ecd3b4e1d8ce1f5c5803a587b8b90d6b5bd048e4888cbef3ed15e47cff4cd6b","observation_id":"80d9cd6b-36de-46fc-9d5f-4d915d320ab6","resolution":{"observed_at":"2026-08-12T18:28:49.296695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-12T05:09:47.944328Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00722","last_updated":"2024-12-01T08:10:04Z","snapshot_observed_at":"2026-08-15T06:03:35.308374Z","submitted_at":"2024-12-01T08:10:04Z","title":"Towards Adaptive Mechanism Activation in Language Agent","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-12T05:09:47.944328Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.00722"},"observation_digest":"sha256:4789d31474f21d07aee89196b418db567ffec2379f8a4aa569d7769598ecce29","observation_id":"03f1aaf6-a946-4a12-8d6b-55ed8d4c73d0","resolution":{"observed_at":"2026-08-12T05:09:47.944328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-11T17:55:13.631034Z","title":"Monte carlo tree search boosts reasoning via iterative pref- erence learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.08442","last_updated":"2024-12-11T15:06:25Z","snapshot_observed_at":"2026-08-16T15:18:46.024318Z","submitted_at":"2024-12-11T15:06:25Z","title":"From Multimodal LLMs to Generalist Embodied Agents: Methods and Lessons","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-11T17:55:13.631034Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.08442"},"observation_digest":"sha256:b1976185d3364bce5aa695f9417b71322d023597fd91b5b40df19d6bd5dd51f6","observation_id":"492cfe2a-9964-483a-9eb5-c6b866a3df65","resolution":{"observed_at":"2026-08-11T17:55:13.631034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-11T12:41:20.732265Z","title":"https://doi.org/10.48550/arXiv.2405.00451","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15282","last_updated":"2024-12-18T15:38:39Z","snapshot_observed_at":"2026-08-17T15:39:28.182814Z","submitted_at":"2024-12-18T15:38:39Z","title":"A Systematic Examination of Preference Learning through the Lens of Instruction-Following","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T12:41:20.732265Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.15282"},"observation_digest":"sha256:ef719248243a3e14c69b5a3f9e70b48791e42ee093e5b8b823732dba4d4158da","observation_id":"9e51b3f5-5634-463f-bc0e-2fa5f74376bd","resolution":{"observed_at":"2026-08-11T12:41:20.732265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-11T10:51:30.375984Z","title":"Monte Carlo tree search boosts reasoning via iterative preference learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16075","last_updated":"2024-12-20T17:19:24Z","snapshot_observed_at":"2026-08-19T21:45:02.859666Z","submitted_at":"2024-12-20T17:19:24Z","title":"Formal Mathematical Reasoning: A New Frontier in AI","version":1},"reference_index":204,"source":"pdf_text","source_observed_at":"2026-08-11T10:51:30.375984Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.16075"},"observation_digest":"sha256:8514dcccd19a84d83ccea17ad412cae0053e81193ff65fa53dc36b0ede7e268b","observation_id":"5d1afb34-b79c-41d5-a450-4c01cd1a1143","resolution":{"observed_at":"2026-08-11T10:51:30.375984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-11T05:58:55.177931Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16964","last_updated":"2024-12-24T11:43:25Z","snapshot_observed_at":"2026-08-17T16:43:01.758405Z","submitted_at":"2024-12-22T10:49:27Z","title":"System-2 Mathematical Reasoning via Enriched Instruction Tuning","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-11T05:58:55.177931Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.16964"},"observation_digest":"sha256:7b2af8e3ea7255724a1dead4c40e4807a551940ab2d29692034414e91637ecfb","observation_id":"f562407d-f349-4e75-ab5c-e7e219e36a82","resolution":{"observed_at":"2026-08-11T05:58:55.177931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-11T05:32:05.157765Z","title":"P.; Kawaguchi, K.; and Shieh, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.17397","last_updated":"2024-12-23T08:51:48Z","snapshot_observed_at":"2026-08-18T09:48:46.475998Z","submitted_at":"2024-12-23T08:51:48Z","title":"Towards Intrinsic Self-Correction Enhancement in Monte Carlo Tree Search Boosted Reasoning via Iterative Preference Learning","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T05:32:05.157765Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.17397"},"observation_digest":"sha256:3b749a1b1247e22a98360a06dcf3fbbd882cf65115d02f24a37937a4285b694e","observation_id":"8af40692-2384-4830-b52c-7e7b4d258317","resolution":{"observed_at":"2026-08-11T05:32:05.157765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-11T04:53:11.851150Z","title":"P., Kawaguchi, K., and Shieh, M","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.18319","last_updated":"2024-12-31T07:41:30Z","snapshot_observed_at":"2026-08-15T19:40:30.225099Z","submitted_at":"2024-12-24T10:07:51Z","title":"Mulberry: Empowering MLLM with o1-like Reasoning and Reflection via Collective Monte Carlo Tree Search","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T04:53:11.851150Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2412.18319"},"observation_digest":"sha256:5dcd59210999ace50e9a5429cab53fb53b505f17f7f41e733b3fa46607ca6088","observation_id":"88b8f79a-7ea5-42ed-aba9-38b42bd20d11","resolution":{"observed_at":"2026-08-11T04:53:11.851150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-10T22:37:07.251016Z","title":"P.; Kawaguchi, K.; and Shieh, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01478","last_updated":"2025-01-02T12:09:17Z","snapshot_observed_at":"2026-08-15T14:59:24.296408Z","submitted_at":"2025-01-02T12:09:17Z","title":"Enhancing Reasoning through Process Supervision with Monte Carlo Tree Search","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-10T22:37:07.251016Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2501.01478"},"observation_digest":"sha256:420a0b795c2c66af8080914ae00f3a0ae8b4d3a83acf5cfc161f87b89a0a74c0","observation_id":"454a3d22-e998-4677-8a83-aa9d8470165a","resolution":{"observed_at":"2026-08-10T22:37:07.251016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-10T20:35:59.671117Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07861","last_updated":"2025-01-14T05:56:26Z","snapshot_observed_at":"2026-08-15T03:54:54.866456Z","submitted_at":"2025-01-14T05:56:26Z","title":"ReARTeR: Retrieval-Augmented Reasoning with Trustworthy Process Rewarding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T20:35:59.671117Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2501.07861"},"observation_digest":"sha256:64e90eac1ebc5343873a959fb3093b90eb6df54c6362a70a1364df1c65974dcb","observation_id":"fc1a4bc7-b4e5-41a8-8d5c-28877f071ae2","resolution":{"observed_at":"2026-08-10T20:35:59.671117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-10T20:10:34.939845Z","title":"P., Kawaguchi, K., and Shieh, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09368","last_updated":"2025-08-11T11:28:39Z","snapshot_observed_at":"2026-08-10T21:23:09.138738Z","submitted_at":"2025-01-16T08:27:40Z","title":"Aligning Instruction Tuning with Pre-training","version":4},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-10T20:10:34.939845Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2501.09368"},"observation_digest":"sha256:9f39bb197bc876b3aa248d97e0c4bf1179a7eac00735e3b8c795ab4260e99285","observation_id":"ce697286-0ca5-40b7-afc7-b01a640d9510","resolution":{"observed_at":"2026-08-10T20:10:34.939845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2501.09686","last_updated":"2025-01-23T08:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T17:37:58Z","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","version":3},"reference_index":167,"source":"pdf_text","source_observed_at":"2026-05-15T21:20:59.128986Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2501.09686"},"observation_digest":"sha256:41f6cc68f88b9ccd30c5d8c501b72a2d7c060c87c59cf39c14f7927b2d7c0e2c","observation_id":"86487eec-c682-434f-a033-314558ab702d","resolution":{"observed_at":"2026-05-15T21:20:59.372426Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2501.09732","last_updated":"2025-01-16T18:30:37Z","snapshot_observed_at":"2026-08-16T15:15:38.321253Z","submitted_at":"2025-01-16T18:30:37Z","title":"Inference-Time Scaling for Diffusion Models beyond Scaling Denoising Steps","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-20T11:45:17.473970Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2501.09732"},"observation_digest":"sha256:11f5ff2cb35177e84c99f502d0662cd253f8f712e0112699edd4b96fb065eae5","observation_id":"ae83e03f-ef37-4ed0-b88f-b93fdb94aeee","resolution":{"observed_at":"2026-05-20T11:45:17.625495Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2502.00955","last_updated":"2026-04-24T08:45:39Z","snapshot_observed_at":"2026-08-15T02:21:16.079524Z","submitted_at":"2025-02-02T23:20:16Z","title":"Efficient Multi-Agent System Training with Data Influence-Oriented Tree Search","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-23T04:06:23.521344Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.00955"},"observation_digest":"sha256:ec6f4479e6716a4639c883e3c30c115ac82d3a2a5009259e42283b11fc6a14ca","observation_id":"7f15cafc-3d7b-4b6b-ba01-2abf8ce793f2","resolution":{"observed_at":"2026-05-23T04:07:30.543382Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-09T14:49:08.851523Z","title":"3 Yu, F., Gao, A., and Wang, B","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.01633","last_updated":"2025-06-25T15:31:17Z","snapshot_observed_at":"2026-08-16T09:09:35.076144Z","submitted_at":"2025-02-03T18:59:01Z","title":"Adversarial Reasoning at Jailbreaking Time","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-09T14:49:08.851523Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.01633"},"observation_digest":"sha256:c04698f6f84ff4e6bb8f8fadfdb79cba9bb9476cd8e3f36a72f06de6552e2f0b","observation_id":"ca48cf8c-0bac-4244-8be4-0fafc950c87c","resolution":{"observed_at":"2026-08-09T14:49:08.851523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-09T17:37:40.802953Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01694","last_updated":"2025-03-01T10:27:24Z","snapshot_observed_at":"2026-08-15T11:22:52.429751Z","submitted_at":"2025-02-02T18:19:14Z","title":"Metastable Dynamics of Chain-of-Thought Reasoning: Provable Benefits of Search, RL and Distillation","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-09T17:37:40.802953Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.01694"},"observation_digest":"sha256:a29876a59a03dfa619a4649739a44de2599f9d1ce894a5631b43bcf9da7854c8","observation_id":"77bd2389-0d2d-41bf-8f95-2dc1e2df1484","resolution":{"observed_at":"2026-08-09T17:37:40.802953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-09T13:25:52.121741Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02095","last_updated":"2025-05-20T12:35:46Z","snapshot_observed_at":"2026-08-13T14:08:47.184627Z","submitted_at":"2025-02-04T08:25:17Z","title":"LongDPO: Unlock Better Long-form Generation Abilities for LLMs via Critique-augmented Stepwise Information","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-09T13:25:52.121741Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.02095"},"observation_digest":"sha256:bae2cd4722854a6fb346a3ed096d57cb405cb875550d559a81900f02dd0dd0bb","observation_id":"7c5120fd-060a-47a6-a75e-7e2775365954","resolution":{"observed_at":"2026-08-09T13:25:52.121741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-09T11:20:31.925154Z","title":"P., Kawaguchi, K., and Shieh, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06813","last_updated":"2025-02-04T22:08:20Z","snapshot_observed_at":"2026-08-19T15:43:28.621336Z","submitted_at":"2025-02-04T22:08:20Z","title":"Policy Guided Tree Search for Enhanced LLM Reasoning","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-09T11:20:31.925154Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.06813"},"observation_digest":"sha256:a9cfd719f206a86d6b47b00064ea1e9ee7dd9513513389f24825efcb3138e8ff","observation_id":"54b0c245-e256-405e-8ae9-67f8c9c35f16","resolution":{"observed_at":"2026-08-09T11:20:31.925154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-10T17:37:46.149210Z","title":"Y., Lillicrap, T","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.15710","last_updated":"2025-01-21T11:32:39Z","snapshot_observed_at":"2026-08-15T03:54:19.967124Z","submitted_at":"2025-01-21T11:32:39Z","title":"The Process of Categorical Clipping at the Core of the Genesis of Concepts in Synthetic Neural Cognition","version":1},"reference_index":148,"source":"pdf_text","source_observed_at":"2026-08-10T17:37:46.149210Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.15710"},"observation_digest":"sha256:a26882a44f947ef2f5f6295b2728690f09999bb9c679447445038d3e4040d9f2","observation_id":"f123c1d2-b171-4acc-94cb-4186e2f1a67d","resolution":{"observed_at":"2026-08-10T17:37:46.149210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2502.17419","last_updated":"2025-06-25T02:24:46Z","snapshot_observed_at":"2026-08-20T08:13:05.630967Z","submitted_at":"2025-02-24T18:50:52Z","title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","version":6},"reference_index":150,"source":"pdf_text","source_observed_at":"2026-05-13T01:36:23.845366Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2502.17419"},"observation_digest":"sha256:1b7be457cf29c0ef04c2ab26c2611b85e2f84eca8084c46dadea2489ecd5a6c7","observation_id":"e7b5c070-3cd7-411a-8e25-612bae7b00b3","resolution":{"observed_at":"2026-05-13T01:36:24.137693Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2503.07536","last_updated":"2025-03-11T03:32:59Z","snapshot_observed_at":"2026-08-18T16:51:32.000170Z","submitted_at":"2025-03-10T17:04:14Z","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-16T15:15:46.255296Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2503.07536"},"observation_digest":"sha256:ddf3b397cd701dc29ecaef273f16f9b8cc182774f2c7617560e9fdb17766af4d","observation_id":"f0e36665-2446-416b-a711-e2d2184c22d7","resolution":{"observed_at":"2026-05-16T15:15:46.329142Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2504.02181","last_updated":"2026-04-22T01:49:17Z","snapshot_observed_at":"2026-07-30T09:24:14.725185Z","submitted_at":"2025-04-02T23:51:27Z","title":"A Survey of Scaling in Large Language Model Reasoning","version":2},"reference_index":231,"source":"pdf_text","source_observed_at":"2026-05-22T21:20:07.238992Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2504.02181"},"observation_digest":"sha256:4f58509ae2312f6cdc1db686ac9dcf4d8103980e6e923dd04625de521c8550be","observation_id":"d344b008-9029-4da1-81dc-eeaff6a905f7","resolution":{"observed_at":"2026-05-22T21:22:09.274394Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2504.11536","last_updated":"2025-04-17T16:46:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-15T18:10:22Z","title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T18:42:39.023650Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2504.11536"},"observation_digest":"sha256:04694079203e59bf65b7f98ec97c5f50495e4597dce2b232434bf2a5423e052c","observation_id":"1c7c246d-3af2-4f84-bb1f-362cefc559c0","resolution":{"observed_at":"2026-05-13T18:42:39.110987Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-16T12:16:31.845707Z","title":"Monte carlo tree search boosts reasoning via iterative prefer- ence learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.13143","last_updated":"2025-04-17T17:51:59Z","snapshot_observed_at":"2026-08-18T07:23:58.831408Z","submitted_at":"2025-04-17T17:51:59Z","title":"$\\texttt{Complex-Edit}$: CoT-Like Instruction Generation for Complexity-Controllable Image Editing Benchmark","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T12:16:31.845707Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2504.13143"},"observation_digest":"sha256:9e98cfbda3023712d5a54d8dd658b941b75c2660ca69085dfd6dec8bd3cff69e","observation_id":"805bb766-d5da-4d30-91a4-13e94bd898b0","resolution":{"observed_at":"2026-08-16T12:16:31.845707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2504.13818","last_updated":"2026-04-22T00:26:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-18T17:49:55Z","title":"Not All Rollouts are Useful: Down-Sampling Rollouts in LLM Reinforcement Learning","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-22T18:46:11.571831Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2504.13818"},"observation_digest":"sha256:08f4349f31baa9d8be7ca27019e4822ace8f66a144cd92fe945ad9bca1a85b3a","observation_id":"deaae569-f1a2-4a1f-9d4a-da39a1702705","resolution":{"observed_at":"2026-05-22T18:46:57.117383Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-16T05:12:18.622009Z","title":"arXiv preprint arXiv:2405.00451 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.21277","last_updated":"2025-05-21T06:08:31Z","snapshot_observed_at":"2026-08-20T10:56:49.638012Z","submitted_at":"2025-04-30T03:14:28Z","title":"Reinforced MLLM: A Survey on RL-Based Reasoning in Multimodal Large Language Models","version":2},"reference_index":122,"source":"pdf_text","source_observed_at":"2026-08-16T05:12:18.622009Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2504.21277"},"observation_digest":"sha256:b362763424d027e91e537d598a1e7491e0cff2b10e0f054c3d1333279196d340","observation_id":"804c29ac-bac8-4d13-83af-ec00d9c83727","resolution":{"observed_at":"2026-08-16T05:12:18.622009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-15T21:45:35.115772Z","title":"”Monte carlo tree search boosts reasoning via iterative preference learning.” arXiv preprint arXiv:2405.00451 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.09029","last_updated":"2025-05-13T23:56:12Z","snapshot_observed_at":"2026-08-20T11:05:33.705067Z","submitted_at":"2025-05-13T23:56:12Z","title":"Monte Carlo Beam Search for Actor-Critic Reinforcement Learning in Continuous Control","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T21:45:35.115772Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.09029"},"observation_digest":"sha256:52c6b240ff6dc96345ca157daf4849c6e97020115a873bbf28054f65e3f8c0ed","observation_id":"ae813dbc-8b2a-44ac-b6b5-de339479dfe9","resolution":{"observed_at":"2026-08-15T21:45:35.115772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-15T21:44:05.867538Z","title":"Monte Carlo tree search boosts reasoning via iterative preference learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.09082","last_updated":"2025-05-14T02:35:47Z","snapshot_observed_at":"2026-08-18T09:48:48.576794Z","submitted_at":"2025-05-14T02:35:47Z","title":"CEC-Zero: Chinese Error Correction Solution Based on LLM","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T21:44:05.867538Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.09082"},"observation_digest":"sha256:80ec9be56eb2892618f3bde3ff8359a9782930538cdf6255cef5903557004249","observation_id":"4f91cc1d-8841-40dd-9af8-9bcdc9cc5114","resolution":{"observed_at":"2026-08-15T21:44:05.867538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-15T20:16:49.367478Z","title":"Yang, J., Jimenez, C","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.13652","last_updated":"2025-05-19T18:50:15Z","snapshot_observed_at":"2026-08-19T14:31:54.280327Z","submitted_at":"2025-05-19T18:50:15Z","title":"Guided Search Strategies in Non-Serializable Environments with Applications to Software Engineering Agents","version":1},"reference_index":451,"source":"pdf_text","source_observed_at":"2026-08-15T20:16:49.367478Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.13652"},"observation_digest":"sha256:0046c9e621410a4e35e128bf4ec82abe0161d6beb3a56a08a1c294a2a83495c6","observation_id":"41751754-982f-4da2-806c-b400fd83650d","resolution":{"observed_at":"2026-08-15T20:16:49.367478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T15:06:01.832895Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16312","last_updated":"2026-06-22T14:23:30Z","snapshot_observed_at":"2026-08-18T18:46:15.782551Z","submitted_at":"2025-05-22T07:07:43Z","title":"EquivPruner: Boosting Efficiency and Quality in LLM-Based Search via Action Pruning","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T15:06:01.832895Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.16312"},"observation_digest":"sha256:b6efd865380d443086298c7a463e20114c4e6b726cb6d55eb348fa74a373aaf1","observation_id":"3996829b-ccc4-498e-994f-8efd91710467","resolution":{"observed_at":"2026-08-07T15:06:01.832895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T14:42:27.461851Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18065","last_updated":"2025-05-23T16:12:12Z","snapshot_observed_at":"2026-08-18T04:15:36.589973Z","submitted_at":"2025-05-23T16:12:12Z","title":"Reward Model Generalization for Compute-Aware Test-Time Reasoning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:42:27.461851Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.18065"},"observation_digest":"sha256:4480019fed5dd0f38e787bb360238f631d8c8ea92bf3e0d81634a99942f30705","observation_id":"b0e8a27f-7ab7-4583-9ee6-bbe4a596520d","resolution":{"observed_at":"2026-08-07T14:42:27.461851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T14:40:42.491053Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18149","last_updated":"2025-05-23T17:57:43Z","snapshot_observed_at":"2026-08-15T17:40:04.118735Z","submitted_at":"2025-05-23T17:57:43Z","title":"First Finish Search: Efficient Test-Time Scaling in Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:40:42.491053Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.18149"},"observation_digest":"sha256:195bb96acf818dc955c5862ee4d67a0734a1c7a4fd363fc89cceb146da8b1088","observation_id":"aab6bc26-c491-4c70-a393-298260d74354","resolution":{"observed_at":"2026-08-07T14:40:42.491053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T14:22:59.566653Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19100","last_updated":"2025-05-25T11:33:08Z","snapshot_observed_at":"2026-08-18T09:48:48.107281Z","submitted_at":"2025-05-25T11:33:08Z","title":"ASPO: Adaptive Sentence-Level Preference Optimization for Fine-Grained Multimodal Reasoning","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T14:22:59.566653Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.19100"},"observation_digest":"sha256:0f581fd0140c3eed3dde5d3d5fb65abc5330e2ab2cd9efe10304d0098c0f09c4","observation_id":"6ec0619b-14a4-4f90-832b-942fe44fa2fa","resolution":{"observed_at":"2026-08-07T14:22:59.566653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T14:12:10.962028Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19683","last_updated":"2025-05-26T08:44:53Z","snapshot_observed_at":"2026-08-13T22:06:58.555085Z","submitted_at":"2025-05-26T08:44:53Z","title":"Large Language Models for Planning: A Comprehensive and Systematic Survey","version":1},"reference_index":288,"source":"pdf_text","source_observed_at":"2026-08-07T14:12:10.962028Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.19683"},"observation_digest":"sha256:1c7e005458505def7176df0be174e07f7a1eeb4b5200dcbe064028fa9ee4e6cb","observation_id":"d28e2eda-1c8d-4a34-b6ab-382e6809f35a","resolution":{"observed_at":"2026-08-07T14:12:10.962028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T14:38:08.638944Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20325","last_updated":"2025-05-23T18:19:09Z","snapshot_observed_at":"2026-08-20T19:20:54.250799Z","submitted_at":"2025-05-23T18:19:09Z","title":"Guided by Gut: Efficient Test-Time Scaling with Reinforced Intrinsic Confidence","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:38:08.638944Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2505.20325"},"observation_digest":"sha256:1a425438ef1e3fc71697dd6a2cd5eb0a29dbbe04645d8396890a8fd1a1f72b7d","observation_id":"9bf720c4-d531-484a-9cea-ec5a7b6a8677","resolution":{"observed_at":"2026-08-07T14:38:08.638944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T12:14:21.204522Z","title":"analysis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00189","last_updated":"2025-05-30T19:59:44Z","snapshot_observed_at":"2026-08-14T08:33:28.076999Z","submitted_at":"2025-05-30T19:59:44Z","title":"Control-R: Towards controllable test-time scaling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:14:21.204522Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.00189"},"observation_digest":"sha256:f00a834a46919fb0c55bb51679480b97c51007a13706a77b7b6dcb462ef2b803","observation_id":"53db4d41-a0ef-45ab-a84d-865bbc38514c","resolution":{"observed_at":"2026-08-07T12:14:21.204522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T10:42:41.085938Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04611","last_updated":"2025-06-05T04:02:17Z","snapshot_observed_at":"2026-08-09T14:45:17.460081Z","submitted_at":"2025-06-05T04:02:17Z","title":"Revisiting Test-Time Scaling: A Survey and a Diversity-Aware Method for Efficient Reasoning","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-07T10:42:41.085938Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.04611"},"observation_digest":"sha256:b25a52623957b117667255a5141eb1dc7d510d8f79816a8c968ee08048566bb2","observation_id":"a09a2f29-2492-42e0-b9f1-000907bd094a","resolution":{"observed_at":"2026-08-07T10:42:41.085938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T10:28:31.403312Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning.arXiv preprint arXiv:2405.00451,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05256","last_updated":"2025-06-06T02:38:39Z","snapshot_observed_at":"2026-08-17T10:46:27.297066Z","submitted_at":"2025-06-05T17:17:05Z","title":"Just Enough Thinking: Efficient Reasoning with Adaptive Length Penalties Reinforcement Learning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:31.403312Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.05256"},"observation_digest":"sha256:74f4096dcf326540e69e971cd6abc6d6ab6a3506236b15872e4c2e7cc9bebe2d","observation_id":"f65d5de8-86c7-4087-8689-c8d1d2d00d61","resolution":{"observed_at":"2026-08-07T10:28:31.403312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T11:00:54.079028Z","title":"Monte carlo tree search boosts reasoning via iterative pref- erence learning.arXiv preprint arXiv:2405.00451, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06366","last_updated":"2025-06-12T10:22:01Z","snapshot_observed_at":"2026-08-09T20:26:32.963470Z","submitted_at":"2025-06-04T08:12:32Z","title":"AI Agent Behavioral Science","version":3},"reference_index":172,"source":"pdf_text","source_observed_at":"2026-08-07T11:00:54.079028Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.06366"},"observation_digest":"sha256:1e5bebb946ba7685d96d6c7c64d58053cc83fea700241b2384aebd0cd9661a3a","observation_id":"8f17ab33-77d9-4642-8147-261668aec8db","resolution":{"observed_at":"2026-08-07T11:00:54.079028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T06:01:49.657996Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning.arXiv preprint arXiv:2405.00451, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06470","last_updated":"2025-06-06T18:55:16Z","snapshot_observed_at":"2026-08-07T05:53:54.045201Z","submitted_at":"2025-06-06T18:55:16Z","title":"SIGMA: Refining Large Language Model Reasoning via Sibling-Guided Monte Carlo Augmentation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T06:01:49.657996Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.06470"},"observation_digest":"sha256:e089f107e9fcfc06cb1c34a34e45907fd50c406eef18e3b81823406c29266908","observation_id":"3f1b2dee-12e0-4eea-ac87-090838360a58","resolution":{"observed_at":"2026-08-07T06:01:49.657996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T04:40:12.319293Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10128","last_updated":"2025-06-11T19:16:54Z","snapshot_observed_at":"2026-08-16T05:21:00.781496Z","submitted_at":"2025-06-11T19:16:54Z","title":"ViCrit: A Verifiable Reinforcement Learning Proxy Task for Visual Perception in VLMs","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-07T04:40:12.319293Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.10128"},"observation_digest":"sha256:fbd61df20b9096a72f44bd3e643785d5b011dc13403c610b66203f58bef23ee6","observation_id":"52378c17-6432-4034-b1cb-652c1758abaf","resolution":{"observed_at":"2026-08-07T04:40:12.319293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-07T01:08:25.027204Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11902","last_updated":"2025-06-13T15:52:37Z","snapshot_observed_at":"2026-08-20T01:36:36.463891Z","submitted_at":"2025-06-13T15:52:37Z","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T01:08:25.027204Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.11902"},"observation_digest":"sha256:0525fbdebfbc2e33fb9c2bb97828dbd6b239b8dcaa6f79390a80954a1ce7d251","observation_id":"23e05b3c-5c67-4de9-af7d-a2f197b49a0f","resolution":{"observed_at":"2026-08-07T01:08:25.027204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-15T19:08:14.502275Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.17828","last_updated":"2025-07-03T05:12:51Z","snapshot_observed_at":"2026-08-18T11:12:16.166349Z","submitted_at":"2025-06-21T21:49:02Z","title":"Aligning Frozen LLMs by Reinforcement Learning: An Iterative Reweight-then-Optimize Approach","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T19:08:14.502275Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.17828"},"observation_digest":"sha256:6ce8d1d37213c5eb2cac2ffce064fa1fb06fdf86fb28b3ba1a4e6df22828a0b2","observation_id":"5dfb5fc4-6623-405c-9e18-0c96c5b66eea","resolution":{"observed_at":"2026-08-15T19:08:14.502275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-06T22:50:14.576341Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20629","last_updated":"2025-06-25T17:25:02Z","snapshot_observed_at":"2026-08-17T15:08:28.482973Z","submitted_at":"2025-06-25T17:25:02Z","title":"PLoP: Precise LoRA Placement for Efficient Finetuning of Large Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T22:50:14.576341Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.20629"},"observation_digest":"sha256:2799b91415f42c8b1d1038a50b2f8c132b2cefdc28a43da49bb25c2c3df831dd","observation_id":"e070fe5c-7e87-41b8-8a10-5bf39df4fad6","resolution":{"observed_at":"2026-08-06T22:50:14.576341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-06T22:30:15.833229Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21497","last_updated":"2025-06-26T17:26:17Z","snapshot_observed_at":"2026-08-11T07:09:44.547805Z","submitted_at":"2025-06-26T17:26:17Z","title":"Enhancing User Engagement in Socially-Driven Dialogue through Interactive LLM Alignments","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T22:30:15.833229Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.21497"},"observation_digest":"sha256:ac7b6ab6623c7ee99f5cdb6be2749a0b1f8a739f6eca3956e011d416734d761b","observation_id":"adf48413-f128-401a-9f20-b1603e517bd6","resolution":{"observed_at":"2026-08-06T22:30:15.833229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-06T21:37:56.996116Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23678","last_updated":"2025-06-30T10:00:43Z","snapshot_observed_at":"2026-08-17T20:30:32.586061Z","submitted_at":"2025-06-30T10:00:43Z","title":"Interactive Reasoning: Visualizing and Controlling Chain-of-Thought Reasoning in Large Language Models","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:56.996116Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2506.23678"},"observation_digest":"sha256:58657828daa9c706c3faae1dbea064d5987e47e72c00aab1ffff794dc3f1af97","observation_id":"11e6cd9d-3b1c-4b70-ac25-4c9c5de4798b","resolution":{"observed_at":"2026-08-06T21:37:56.996116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-15T18:39:18.997563Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00045","last_updated":"2025-06-23T22:05:21Z","snapshot_observed_at":"2026-08-20T17:33:27.320921Z","submitted_at":"2025-06-23T22:05:21Z","title":"CaughtCheating: Is Your MLLM a Good Cheating Detective? Exploring the Boundary of Visual Perception and Reasoning","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-15T18:39:18.997563Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2507.00045"},"observation_digest":"sha256:a312fe8a6b24843851b453fc2835ae2f6877069920d2fc618e692311df3121ba","observation_id":"e5e80796-9eed-4796-bcae-9ef353d7f30d","resolution":{"observed_at":"2026-08-15T18:39:18.997563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2507.04736","last_updated":"2026-04-10T04:48:11Z","snapshot_observed_at":"2026-08-19T05:55:10.527761Z","submitted_at":"2025-07-07T08:08:20Z","title":"ChipSeek: Optimizing Verilog Generation via EDA-Integrated Reinforcement Learning","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-19T06:48:29.015759Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2507.04736"},"observation_digest":"sha256:8963140adf85af55adf7d3780b6b0b4c0f6d7d1830b166ed0f379340dcd8ec44","observation_id":"1c03327a-0a26-410a-9f02-e0d0947c75a1","resolution":{"observed_at":"2026-05-19T06:52:08.359154Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T23:49:05.025670Z","title":"arXiv preprint arXiv:2405.00451","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.04848","last_updated":"2025-08-06T19:51:29Z","snapshot_observed_at":"2026-08-18T21:37:52.920383Z","submitted_at":"2025-08-06T19:51:29Z","title":"Large Language Models Reasoning Abilities Under Non-Ideal Conditions After RL-Fine-Tuning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T23:49:05.025670Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2508.04848"},"observation_digest":"sha256:a5c4f58fe7ac393ae4c37e6dceff2d279a85018b43db038d054dee01a2cd073d","observation_id":"da8eb2be-757c-408d-9572-55982b3900b1","resolution":{"observed_at":"2026-08-05T23:49:05.025670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T15:38:55.207976Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19689","last_updated":"2025-08-27T08:52:47Z","snapshot_observed_at":"2026-08-07T07:48:25.655053Z","submitted_at":"2025-08-27T08:52:47Z","title":"Building Task Bots with Self-learning for Enhanced Adaptability, Extensibility, and Factuality","version":1},"reference_index":198,"source":"arxiv_source","source_observed_at":"2026-08-05T15:38:55.207976Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2508.19689"},"observation_digest":"sha256:2bfa553bd5494d757c84ca6d5ce543e13a78fb85850124a2d8c06dee2c4aeed1","observation_id":"06bec0d1-228a-4eb2-8616-7e5318d67b2e","resolution":{"observed_at":"2026-08-05T15:38:55.207976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T13:24:39.898650Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00676","last_updated":"2025-08-31T03:08:02Z","snapshot_observed_at":"2026-08-14T04:31:24.300394Z","submitted_at":"2025-08-31T03:08:02Z","title":"LLaVA-Critic-R1: Your Critic Model is Secretly a Strong Policy Model","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-05T13:24:39.898650Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2509.00676"},"observation_digest":"sha256:5a08dccece6c1daa136278c6754315335d7035bc9b0a8a94614d9339742e9f14","observation_id":"3ed8bf8c-fce5-4454-a622-c5bef9720374","resolution":{"observed_at":"2026-08-05T13:24:39.898650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2509.21743","last_updated":"2026-03-31T22:32:08Z","snapshot_observed_at":"2026-08-17T12:57:06.767729Z","submitted_at":"2025-09-26T01:17:35Z","title":"Retrieval-of-Thought: Efficient Reasoning via Reusing Thoughts","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-18T13:42:07.883909Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2509.21743"},"observation_digest":"sha256:f3e1eaae579312603b9b561d4b5e9e915e2ddff6f4c503fc8605892cc1c99b13","observation_id":"aa814ac9-bc2e-474d-9163-cfea9b35fa5e","resolution":{"observed_at":"2026-05-18T13:42:38.694348Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-04T14:44:39.878932Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.23946","last_updated":"2026-06-22T03:40:52Z","snapshot_observed_at":"2026-08-12T21:22:31.257256Z","submitted_at":"2025-09-28T15:48:40Z","title":"Explore-Execute Chain: Towards an Efficient Structured Reasoning Paradigm","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-04T14:44:39.878932Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2509.23946"},"observation_digest":"sha256:ef1c059c7089f100721e531d2b791108ed685b645a4d0fb3d3101ce802be97ec","observation_id":"576f190c-e985-4613-8133-602f0e0edbac","resolution":{"observed_at":"2026-08-04T14:44:39.878932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2510.08592","last_updated":"2026-05-09T15:18:45Z","snapshot_observed_at":"2026-08-11T12:02:28.088257Z","submitted_at":"2025-10-04T20:01:21Z","title":"Less Diverse, Less Safe: The Indirect But Pervasive Risk of Test-Time Scaling in Large Language Models","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T09:53:57.765473Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2510.08592"},"observation_digest":"sha256:3de6003a9e8d82ebc37fb931e8039401e5392014df1ab1c87832650cd1ba0988","observation_id":"fbf825b4-52e4-4b83-8952-4c4bae846dfd","resolution":{"observed_at":"2026-05-18T09:56:13.094934Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2510.13786","last_updated":"2025-10-15T17:43:03Z","snapshot_observed_at":"2026-08-18T16:44:13.851821Z","submitted_at":"2025-10-15T17:43:03Z","title":"The Art of Scaling Reinforcement Learning Compute for LLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T16:29:13.954029Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2510.13786"},"observation_digest":"sha256:8a65c86dbe2f97b63c00b583edb068eee3f8a2f546f605c339d70e605ceb1fd5","observation_id":"0f043711-7720-401a-a758-a72b39b21b18","resolution":{"observed_at":"2026-05-16T16:29:14.038199Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-04T08:11:52.628741Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.22228","last_updated":"2026-07-20T17:30:01Z","snapshot_observed_at":"2026-08-18T05:20:10.037311Z","submitted_at":"2025-10-25T09:22:22Z","title":"When Fewer Layers Break More Chains: Layer Pruning Harms Test-Time Scaling in LLMs","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-04T08:11:52.628741Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2510.22228"},"observation_digest":"sha256:d08a3c62bfbbeb581aaf4287fe9b14be251856f491d56bb3e665466a350c5386","observation_id":"b56b40c4-68a7-4348-99d7-7eb4484f2f73","resolution":{"observed_at":"2026-08-04T08:11:52.628741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2512.07461","last_updated":"2026-05-14T12:43:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-08T11:39:43Z","title":"Native Parallel Reasoner: Reasoning in Parallelism via Self-Distilled Reinforcement Learning","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T01:11:26.411893Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2512.07461"},"observation_digest":"sha256:b034a3ba76a168a2a775799baf9fef3a34d1ae2c5b4fc7601f41855f15413747","observation_id":"1336d905-1baf-45fd-9289-70be4433e537","resolution":{"observed_at":"2026-05-17T01:13:47.957447Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-03T17:15:17.701284Z","title":"Monte carlo tree search boosts reasoning via iterative prefer- ence learning.arXiv preprint arXiv:2405.00451, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.10414","last_updated":"2026-05-30T03:42:16Z","snapshot_observed_at":"2026-08-15T18:36:58.856582Z","submitted_at":"2025-12-11T08:27:02Z","title":"Boosting RL-Based Visual Reasoning with Selective Adversarial Entropy Intervention","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-03T17:15:17.701284Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2512.10414"},"observation_digest":"sha256:ae00a757903050f5005eb7e0c8c0b723a1fbb50e24af03eef7984e7db3aa9edc","observation_id":"f41c982a-01ce-450e-8365-cea84e37ddb8","resolution":{"observed_at":"2026-08-03T17:15:17.701284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-02T23:50:37.479610Z","title":"P., Kawaguchi, K., and Shieh, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.12586","last_updated":"2026-05-27T16:01:59Z","snapshot_observed_at":"2026-08-15T07:57:20.327359Z","submitted_at":"2026-02-13T03:56:22Z","title":"Can I Have Your Order? Monte-Carlo Tree Search for Slot Filling Ordering in Diffusion Language Models","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-02T23:50:37.479610Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2602.12586"},"observation_digest":"sha256:979f9c659808502e856f624d6c4b31460d99ea9d448d19fba7daa60c4bf59e6c","observation_id":"1427f7f5-f06b-4302-885a-0aeb295ecd74","resolution":{"observed_at":"2026-08-02T23:50:37.479610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.00323","last_updated":"2026-05-01T01:03:05Z","snapshot_observed_at":"2026-08-17T12:57:05.750456Z","submitted_at":"2026-05-01T01:03:05Z","title":"Online Self-Calibration Against Hallucination in Vision-Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-09T20:20:27.931679Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.00323"},"observation_digest":"sha256:8449318960a1fdad033441e196b74827bab74c8a6905716d26bcad09d21782f8","observation_id":"8b0a4382-6d8c-49de-bae7-f6361f8e2a36","resolution":{"observed_at":"2026-05-11T15:16:10.758067Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.04831","last_updated":"2026-05-06T12:28:17Z","snapshot_observed_at":"2026-08-15T16:16:01.364447Z","submitted_at":"2026-05-06T12:28:17Z","title":"StoryAlign: Evaluating and Training Reward Models for Story Generation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-08T17:29:13.549559Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.04831"},"observation_digest":"sha256:8c866de34e1e2626e4e8bdc58c17a6e643fbf68692b6871edbb4e4b662981acc","observation_id":"68714fb5-68aa-476d-9c59-4a3b6fff7f38","resolution":{"observed_at":"2026-05-11T17:31:07.625809Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.06200","last_updated":"2026-05-07T13:09:31Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T13:09:31Z","title":"A$^2$TGPO: Agentic Turn-Group Policy Optimization with Adaptive Turn-level Clipping","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-08T10:41:46.675257Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.06200"},"observation_digest":"sha256:8044c7e35eba672a6360396792c958dcfebdd23d730c4d452fa0b63f350d6574","observation_id":"c8ef7e11-e612-4411-85d7-be0817289dcd","resolution":{"observed_at":"2026-05-11T19:56:08.803918Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.08057","last_updated":"2026-05-08T17:44:15Z","snapshot_observed_at":"2026-08-11T07:32:29.834666Z","submitted_at":"2026-05-08T17:44:15Z","title":"CA-SQL: Complexity-Aware Inference Time Reasoning for Text-to-SQL via Exploration and Compute Budget Allocation","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-11T02:28:20.366674Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.08057"},"observation_digest":"sha256:d79720e87dac041d20eb60b03d89ce2495d2ff0c9db8582a3e7d901a0204d467","observation_id":"a3040011-7ea3-44ec-a821-3ec480e3f0aa","resolution":{"observed_at":"2026-05-11T03:25:58.162248Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.09492","last_updated":"2026-05-20T15:55:09Z","snapshot_observed_at":"2026-07-06T23:21:35.327066Z","submitted_at":"2026-05-10T11:57:39Z","title":"APCD: Adaptive Path-Contrastive Decoding for Reliable Large Language Model Generation","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-12T05:22:25.475956Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.09492"},"observation_digest":"sha256:affb85c5827a8db695c57367d3566315808af959f00faca40090323352be5822","observation_id":"68c07757-b090-4bc8-8122-a6503e8fca4c","resolution":{"observed_at":"2026-05-12T05:26:25.114868Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.10195","last_updated":"2026-05-14T07:42:56Z","snapshot_observed_at":"2026-08-03T09:42:12.047946Z","submitted_at":"2026-05-11T08:45:17Z","title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-12T03:51:52.375703Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.10195"},"observation_digest":"sha256:27e050c3fe51ea2f1b57ffbef5e55738e7f00eb316f9c8d435b34ade19b18909","observation_id":"c55c9fe3-ad48-4d46-99ac-8f5548f67e17","resolution":{"observed_at":"2026-05-12T06:51:30.308824Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.10195","last_updated":"2026-05-14T07:42:56Z","snapshot_observed_at":"2026-08-03T09:42:12.047946Z","submitted_at":"2026-05-11T08:45:17Z","title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-15T05:11:32.053440Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.10195"},"observation_digest":"sha256:5900d1933d7e725b7b1f11d0de20ca4609999ffb4758bfcf1800b8185c40d263","observation_id":"57f9ef26-61e7-4a7b-9c54-95760f60ccb7","resolution":{"observed_at":"2026-05-15T05:15:03.377542Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.10913","last_updated":"2026-06-24T17:55:07Z","snapshot_observed_at":"2026-08-13T15:51:48.523223Z","submitted_at":"2026-05-11T17:50:51Z","title":"Shepherd: Enabling Programmable Meta-Agents via Reversible Agentic Execution Traces","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-12T03:29:33.497561Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.10913"},"observation_digest":"sha256:dc1a51956b34e34ca2fd53b2a4669b8cc4a5bcfbca7e7efe7bc4bdf424bb0ffe","observation_id":"15b827cc-a7dd-4a11-9f5c-b797760a1816","resolution":{"observed_at":"2026-05-12T07:16:30.885704Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.10913","last_updated":"2026-06-24T17:55:07Z","snapshot_observed_at":"2026-08-13T15:51:48.523223Z","submitted_at":"2026-05-11T17:50:51Z","title":"Shepherd: Enabling Programmable Meta-Agents via Reversible Agentic Execution Traces","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-30T22:24:26.528532Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.10913"},"observation_digest":"sha256:07e734861106faf40ab9b654e7d8035a35cace4a0cf025859dfd943632aae7f0","observation_id":"48f329f6-4104-4082-8909-01fa1b90e315","resolution":{"observed_at":"2026-06-30T22:25:06.761279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.12384","last_updated":"2026-05-12T16:47:40Z","snapshot_observed_at":"2026-07-06T23:24:03.984179Z","submitted_at":"2026-05-12T16:47:40Z","title":"Scalable Token-Level Hallucination Detection in Large Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T05:49:23.534294Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.12384"},"observation_digest":"sha256:29fd5cb6783608c3a8fca5e2be985ac5d13eb2756bece780a1c9ca471cc41897","observation_id":"64f8b47c-59d7-4d39-a4ba-3ba70e8c492b","resolution":{"observed_at":"2026-05-13T05:52:22.455333Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2605.24326","last_updated":"2026-05-23T01:11:19Z","snapshot_observed_at":"2026-08-18T01:55:53.399683Z","submitted_at":"2026-05-23T01:11:19Z","title":"ScaleAcross Explorer: Exploring Communication Optimization for Scale-Across AI Model Training","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-30T12:57:37.344369Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2605.24326"},"observation_digest":"sha256:189e8637f7e1025838c33d9a62f80c479dd0c7fc9e3f6aeee40fa4d65def4384","observation_id":"6e1d9e44-cb39-4d6f-94dd-d1797095a35d","resolution":{"observed_at":"2026-06-30T13:04:40.483302Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.01667","last_updated":"2026-06-01T04:19:50Z","snapshot_observed_at":"2026-08-03T01:20:45.755599Z","submitted_at":"2026-06-01T04:19:50Z","title":"ATLAS: Agentic Test-time Learning-to-Allocate Scaling","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-28T15:27:28.290178Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.01667"},"observation_digest":"sha256:235fac0886491e8ac2b615ba386f6320151712eac75f9d7854df93048b35280e","observation_id":"bae10265-a6c0-4d2c-bc44-db096746d7aa","resolution":{"observed_at":"2026-07-01T22:26:17.086263Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.05464","last_updated":"2026-06-03T21:43:38Z","snapshot_observed_at":"2026-08-03T10:19:08.677126Z","submitted_at":"2026-06-03T21:43:38Z","title":"Step-by-Step Optimization-like Reasoning in LLMs over Expanding Search Spaces","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-28T05:46:26.938277Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.05464"},"observation_digest":"sha256:73e8bf54cd1f9f4f4a9902a31249db40b88b90bb476679547445ecf32cc4aee5","observation_id":"6e49eddc-22ef-4fce-a20d-44cbeb608b66","resolution":{"observed_at":"2026-07-02T08:46:48.943212Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.08231","last_updated":"2026-06-06T15:39:29Z","snapshot_observed_at":"2026-08-07T16:58:05.550639Z","submitted_at":"2026-06-06T15:39:29Z","title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-06-27T19:36:57.231932Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.08231"},"observation_digest":"sha256:23a351b70733f6cea2c50f3e8f64f4f923c9d743e8347c306b96be3c65cf38d7","observation_id":"f6beadf6-5ee6-45d4-b3d0-40f50864e89c","resolution":{"observed_at":"2026-07-02T21:37:25.354455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.11119","last_updated":"2026-06-09T17:16:03Z","snapshot_observed_at":"2026-08-18T17:54:27.819254Z","submitted_at":"2026-06-09T17:16:03Z","title":"TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-06-27T13:55:35.363377Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.11119"},"observation_digest":"sha256:015d853451382a9579db12b723fd2552aeb8c38a330f48ee5c713171070d74b2","observation_id":"2bc63d21-22b7-432f-b193-237cc353379a","resolution":{"observed_at":"2026-07-03T04:27:36.752547Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.12384","last_updated":"2026-07-31T08:54:09Z","snapshot_observed_at":"2026-08-05T23:10:46.327902Z","submitted_at":"2026-06-10T17:47:07Z","title":"APPO: Agentic Procedural Policy Optimization","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-06-27T10:21:55.485624Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.12384"},"observation_digest":"sha256:54d4f24f4b63c6c61c106385dbc832447f3fa94b79dbc7febd296ed738477969","observation_id":"28117a7f-5f6a-4c38-b65b-444a7c346444","resolution":{"observed_at":"2026-07-03T09:37:49.336402Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-03T02:12:34.483141Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.12384","last_updated":"2026-07-31T08:54:09Z","snapshot_observed_at":"2026-08-05T23:10:46.327902Z","submitted_at":"2026-06-10T17:47:07Z","title":"APPO: Agentic Procedural Policy Optimization","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-03T02:12:34.483141Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.12384"},"observation_digest":"sha256:3c2b5fb260736de525c344ef7f6cffa60314ea180741d41c55cb061633ed57f1","observation_id":"e9977467-00f8-4151-8f7b-7bd97cb7a3cf","resolution":{"observed_at":"2026-08-03T02:12:34.483141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.13316","last_updated":"2026-07-31T08:57:20Z","snapshot_observed_at":"2026-08-13T10:34:20.953214Z","submitted_at":"2026-06-11T13:10:48Z","title":"ReSum: Synergizing LLM Reasoning and Summarization with Reinforcement Learning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-27T06:39:34.199607Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.13316"},"observation_digest":"sha256:d7e766a32fe1296d1709b257becb473c34ece71d0e0acc8439437578faabdd62","observation_id":"be34a2a5-a3d3-41c9-9d02-0bf58118a20d","resolution":{"observed_at":"2026-07-03T15:08:33.369931Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-03T02:12:23.860218Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.13316","last_updated":"2026-07-31T08:57:20Z","snapshot_observed_at":"2026-08-13T10:34:20.953214Z","submitted_at":"2026-06-11T13:10:48Z","title":"ReSum: Synergizing LLM Reasoning and Summarization with Reinforcement Learning","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-03T02:12:23.860218Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.13316"},"observation_digest":"sha256:c0accc3ab71be4b6b84663245444500774f40162b746aaa9bdddcb317797d256","observation_id":"b85efc86-f7b5-42fb-a897-3b8873fc3b52","resolution":{"observed_at":"2026-08-03T02:12:23.860218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.20599","last_updated":"2026-05-19T09:13:21Z","snapshot_observed_at":"2026-08-11T16:25:26.788078Z","submitted_at":"2026-05-19T09:13:21Z","title":"Beyond Fixed Budgets: Characterizing the Inelasticity and Limitations of Tree-of-Thought Reasoning Strategies","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-30T18:38:48.512777Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.20599"},"observation_digest":"sha256:aaef70b64be7c6ebe0bec76cff01a2482a596efc001a74a6a6dc507a19fa5738","observation_id":"824af795-821b-434e-bc1f-1cdba21dd220","resolution":{"observed_at":"2026-06-30T18:44:59.791395Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.21943","last_updated":"2026-06-20T08:20:41Z","snapshot_observed_at":"2026-08-15T21:55:25.625601Z","submitted_at":"2026-06-20T08:20:41Z","title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","version":1},"reference_index":236,"source":"pdf_text","source_observed_at":"2026-06-26T12:15:08.304150Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.21943"},"observation_digest":"sha256:766e91dfdd8114af5abccb0bb359bb18581aee4aee5c2f4f3f6010ddc76e68db","observation_id":"1410f6a1-38bf-4ab8-90ae-624ba2f850b4","resolution":{"observed_at":"2026-07-04T08:09:40.727786Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.25354","last_updated":"2026-06-29T23:57:32Z","snapshot_observed_at":"2026-08-13T15:14:17.866685Z","submitted_at":"2026-06-24T03:42:44Z","title":"Efficient and Trainable Language Model Test-Time Scaling via Local Branch Routing","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-25T21:29:40.564852Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.25354"},"observation_digest":"sha256:6f41a98ad6c55b613270031c1dcb44a210d7331f423c9a48229a5f43e0bb6904","observation_id":"bd4cebbd-1f7e-4e37-97ef-503b2241cf42","resolution":{"observed_at":"2026-07-04T19:20:06.708349Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.25354","last_updated":"2026-06-29T23:57:32Z","snapshot_observed_at":"2026-08-13T15:14:17.866685Z","submitted_at":"2026-06-24T03:42:44Z","title":"Efficient and Trainable Language Model Test-Time Scaling via Local Branch Routing","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-01T06:33:45.824727Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.25354"},"observation_digest":"sha256:7c9a67bf023b0f56acc18fb0f8f5a9ab22e35b4a3da8e0045c254b55718c435f","observation_id":"0d3dd942-0e67-4127-8a6d-21c116e33c71","resolution":{"observed_at":"2026-07-01T06:35:29.541290Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":"2405.00451","doi":"10.48550/arxiv.2405.00451","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"P., Kawaguchi, K., and Shieh, M","venue":"arXiv (Cornell University)","work_id":"c32202fe-3b76-413d-8db1-885ff3480fe6","year":2024},"citing_paper":{"arxiv_id":"2606.31748","last_updated":"2026-06-30T14:38:49Z","snapshot_observed_at":"2026-07-07T00:05:27.130060Z","submitted_at":"2026-06-30T14:38:49Z","title":"Addressing Over-Refusal in LLMs with Competing Rewards","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-07-01T06:59:12.695984Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2606.31748"},"observation_digest":"sha256:7d2179dd5c8ceb09aac720c260188edc3fba2a281d361ea995c14d796dfdd550","observation_id":"f526cda6-e761-4871-83c3-6b5e4f3a5928","resolution":{"observed_at":"2026-07-01T08:55:35.605736Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-07-11T13:53:36.775836Z","title":"arXiv preprint arXiv:2405.00451 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-15T22:59:02.741838Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-07-11T13:53:36.775836Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:442a45f90e00b295b835fa01d42d133a6613093ff12eb173b66448edadbddac9","observation_id":"f5348cad-0dee-43c0-8c6e-a9a67b0c39fe","resolution":{"observed_at":"2026-07-11T13:53:36.775836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-02T08:40:39.787640Z","title":"arXiv preprint arXiv:2405.00451 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-15T22:59:02.741838Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":3},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-02T08:40:39.787640Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:1ac10d2d93ce19753d796e394a93525796bc77b9fd97ca4b934e616e953918f2","observation_id":"611ae85f-0e67-41c6-b984-776c7bf72877","resolution":{"observed_at":"2026-08-02T08:40:39.787640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-02T14:50:17.718586Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning.arXiv preprint arXiv:2405.00451, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16204","last_updated":"2026-05-07T00:40:32Z","snapshot_observed_at":"2026-08-19T19:36:19.217904Z","submitted_at":"2026-05-07T00:40:32Z","title":"Masked Diffusion Language Models are Strong and Steerable Text-Based World Models for Agentic RL","version":1},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-08-02T14:50:17.718586Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2607.16204"},"observation_digest":"sha256:1ce4266dac903cfd21e20527a919c9a057d0b67998a4b8855853cb32f4f4384c","observation_id":"9ab0484c-befc-4fdd-877f-07d23aa35314","resolution":{"observed_at":"2026-08-02T14:50:17.718586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-15T15:43:00.157724Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17823","last_updated":"2026-07-20T11:11:55Z","snapshot_observed_at":"2026-08-17T01:58:36.348085Z","submitted_at":"2026-07-20T11:11:55Z","title":"Theoretical Foundations of $\\max$@$k$ Reinforcement Learning","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-15T15:43:00.157724Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2607.17823"},"observation_digest":"sha256:12e88573b4b2486f67925ff70a5aec02cacf91711a44136a1f1c3926f8d7da77","observation_id":"aaaaf2ff-7240-4b6f-b5d1-0bd167f1baa2","resolution":{"observed_at":"2026-08-15T15:43:00.157724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-12T00:38:25.264770Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08020","last_updated":"2026-08-11T04:35:54Z","snapshot_observed_at":"2026-08-19T14:32:43.550015Z","submitted_at":"2026-08-08T09:00:01Z","title":"Thought-Level Beam Search for Reasoning","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-12T00:38:25.264770Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2608.08020"},"observation_digest":"sha256:523681dd3a394a2e94f74c6f2b090f600f554d345aa735bb80e5851143bf0327","observation_id":"ab2469e5-4b98-4ee1-8904-3346aacf9b3e","resolution":{"observed_at":"2026-08-12T00:38:25.264770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-15T14:30:50.755233Z","title":"Lillicrap, Kenji Kawaguchi, and Michael Shieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08020","last_updated":"2026-08-11T04:35:54Z","snapshot_observed_at":"2026-08-19T14:32:43.550015Z","submitted_at":"2026-08-08T09:00:01Z","title":"Thought-Level Beam Search for Reasoning","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-15T14:30:50.755233Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2608.08020"},"observation_digest":"sha256:d8703ea0ef50bf0c7a7617aaaa936050dc377cc998d659aba32bac4cda9fbefe","observation_id":"42eeda29-89e7-4ce3-853e-7adcccbc75ce","resolution":{"observed_at":"2026-08-15T14:30:50.755233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-12T14:10:45.699834Z","title":"arXiv preprint arXiv:2405.00451 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10928","last_updated":"2026-08-11T13:58:07Z","snapshot_observed_at":"2026-08-17T15:17:05.825896Z","submitted_at":"2026-08-11T13:58:07Z","title":"ThinkRetrieve: Retrieval-Augmented Reasoning Traces for Test-Time Scaling","version":1},"reference_index":166,"source":"arxiv_source","source_observed_at":"2026-08-12T14:10:45.699834Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2608.10928"},"observation_digest":"sha256:2504b5c874b91a9fa39350930e7d1dd39182c073a0c1dd80743a9557d7322c2a","observation_id":"ac3caf52-92a7-4a2a-a78f-cce02a3d2cf7","resolution":{"observed_at":"2026-08-12T14:10:45.699834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00451","snapshot_observed_at":"2026-08-16T00:21:59.252295Z","title":"Monte carlo tree search boosts reasoning via iterative preference learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.12062","last_updated":"2026-08-12T13:48:30Z","snapshot_observed_at":"2026-08-18T16:33:48.146408Z","submitted_at":"2026-08-12T13:48:30Z","title":"Preference Tree Optimization: Enhancing Goal-Oriented Dialogue with Look-Ahead Simulations","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-16T00:21:59.252295Z"},"links":{"cited_paper":"/paper/2405.00451","citing_paper":"/paper/2608.12062"},"observation_digest":"sha256:605380cab9d8c6d35adc2d378504bef5f812fb9abe4daf80ead77cf9b1f86708","observation_id":"42632a1c-3669-4b61-9f89-470f2fcefc82","resolution":{"observed_at":"2026-08-16T00:21:59.252295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2405.00451/citation-record","integrity":"/paper/2405.00451/integrity","json":"/paper/2405.00451/citation-record.json","paper":"/paper/2405.00451"},"outbound":[],"paper":{"arxiv_id":"2405.00451","last_updated":"2024-06-17T22:11:49Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-18T09:48:11.842015Z","submitted_at":"2024-05-01T11:10:24Z","title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 91 inbound Pith citation observations for arXiv:2405.00451."}