{"as_of":"2026-08-07T01:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:80f603bf44ded0d39165ce2eeca3608b2060987e118debb9141f62e7f890a2bd","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T00:48:56.892634Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.17735/citation-record","integrity":"/paper/2606.17735/integrity","json":"/paper/2606.17735/citation-record.json","paper":"/paper/2606.17735"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-04T16:07:24.699834Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"cited_work":{"arxiv_id":"2509.16679","doi":"10.48550/arxiv.2509.16679","metadata_source":"pith","pith_arxiv_id":"2509.16679","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning meets large language models: A survey of advancements and applications across the LLM lifecycle","venue":"cs.CL","work_id":"f4d2b80d-8fa3-41ad-a490-eab600e35f2f","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2509.16679","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:a83cbb97c230e9fb9f414af6e4a0d936f570f64c000864e607660fa5adbe839d","observation_id":"bb33b206-c858-4eb3-88ed-e13ab3645cda","resolution":{"observed_at":"2026-07-03T21:08:58.849123Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"cited_work":{"arxiv_id":"2509.08827","doi":"10.48550/arxiv.2509.08827","metadata_source":"pith","pith_arxiv_id":"2509.08827","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","venue":"cs.CL","work_id":"7618c14b-e527-4268-9926-d4f462ea9925","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2509.08827","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:4e74cd0fe3ecc641a1514164d5d91b02442ca3179efbf9707805a40d0dc00ff8","observation_id":"d5faa1f9-4e02-4910-85cd-3fda3b150e26","resolution":{"observed_at":"2026-07-03T21:08:58.861020Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-07-06T20:30:26.881629Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":"2502.01600","doi":"10.48550/arxiv.2502.01600","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reinforcement learning for long-horizon interactive llm agents","venue":"ArXiv.org","work_id":"7e929792-6a2a-42ff-a1db-763f890d8b4e","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:3057f3ec4e72c6a7b71e0fad1c5063a9a2308b38e7c401e06a6dc8c1c88c5f05","observation_id":"ca03c586-95e5-467b-a080-52bc8ff08e3b","resolution":{"observed_at":"2026-07-03T21:18:58.371419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01296","last_updated":"2025-04-02T01:59:26Z","snapshot_observed_at":"2026-07-30T08:40:23.849282Z","submitted_at":"2025-04-02T01:59:26Z","title":"ThinkPrune: Pruning Long Chain-of-Thought of LLMs via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2504.01296","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.01296","snapshot_observed_at":"2026-07-10T05:16:48.068473Z","title":"ThinkPrune: Pruning Long Chain-of-Thought of LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"9eb97578-ef7d-45eb-b973-842ff7f61c2b","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2504.01296","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:9be0086e60c6dbe8c685f3a120593dedc124c89661d5a323f1df18fed09db1b2","observation_id":"c52240c1-8467-4371-9340-c9db88b9fb3a","resolution":{"observed_at":"2026-07-03T21:18:58.398809Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.06176","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.372516Z","title":"Large language model reasoning failures.arXiv preprint arXiv:2602.06176","venue":null,"work_id":"a812ddd4-acf3-417b-8079-4450d0c51514","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:e2009e23a0af430acb43074924d96075388332b278539851730fa4ae1f4ca914","observation_id":"60250b10-1048-4cf3-a447-b61dc719b8d5","resolution":{"observed_at":"2026-07-03T21:18:58.374020Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.01288","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T20:00:08.461635Z","title":"arXiv preprint arXiv:2602.01288 , year=","venue":null,"work_id":"31d1b378-5b49-421a-96c7-ec3148393196","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:30eabeb775899a717b797207826a6d0195f194dc98edb0f4b2f807194ceed18f","observation_id":"10196b18-3880-4df7-a95e-e48a0fa945f5","resolution":{"observed_at":"2026-07-03T21:18:58.392439Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.16874","last_updated":"2026-05-16T08:33:31Z","snapshot_observed_at":"2026-08-01T16:12:13.947960Z","submitted_at":"2026-05-16T08:33:31Z","title":"Reasoning Can Be Restored by Correcting a Few Decision Tokens","version":1},"cited_work":{"arxiv_id":"2605.16874","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.16874","snapshot_observed_at":"2026-07-04T10:29:44.959212Z","title":"Reasoning Can Be Restored by Correcting a Few Decision Tokens","venue":"cs.AI","work_id":"771ccecc-a586-487a-aa25-60d6ae5916db","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2605.16874","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:c5ad06b64e0ee61e8b4b354f81313093239768b2ee8a582ed288928ec150ec85","observation_id":"ce81ecab-1249-4d57-ba80-818dd8c6825b","resolution":{"observed_at":"2026-07-03T21:18:58.340911Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19853","last_updated":"2024-06-28T11:52:53Z","snapshot_observed_at":"2026-08-03T12:53:21.818536Z","submitted_at":"2024-06-28T11:52:53Z","title":"YuLan: An Open-source Large Language Model","version":1},"cited_work":{"arxiv_id":"2406.19853","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.19853","snapshot_observed_at":"2026-07-03T21:18:58.385004Z","title":"Yulan: An open-source large language model.arXiv preprint arXiv:2406.19853,","venue":null,"work_id":"e4e89b90-f590-4e3b-8b3a-d461dc34f362","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2406.19853","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:eb9ecbeaeb6e64b05faecec6e8601acac6cfb6121962587b1392c611b88561c5","observation_id":"4bf21333-ba45-47d0-8aa3-d524641dcc6a","resolution":{"observed_at":"2026-07-03T21:18:58.386862Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.16790","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.381583Z","title":"org/abs/2603.16790","venue":null,"work_id":"ec17a712-1ae1-468e-b61b-8adfd768c5b2","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:247729bd7a72556e7d35d15f49a6a599ab2a249bdb55f97b5b399eca4f369c5d","observation_id":"3e9b128a-5262-44cf-a893-034adde9778d","resolution":{"observed_at":"2026-07-03T21:18:58.383757Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.16733","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.383237Z","title":"Iquest-coder-v1 technical report.arXiv preprint arXiv:2603.16733, 2026b","venue":null,"work_id":"0052477f-d3f9-402e-bfbd-7bbb26c22b29","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:609262d726273e19e3004099934f749ab375454577045d165adcbc2bde2b03d2","observation_id":"5eb652c5-a99a-4dd6-8538-e5b53434f88c","resolution":{"observed_at":"2026-07-03T21:18:58.384774Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:649ad321a9ef59f90529f506a8b8debad7b3643b4c34dfcdd8ea3850a628a8dd","observation_id":"ea62d952-3925-4f51-9dc8-d2fe78c4813d","resolution":{"observed_at":"2026-07-03T21:18:58.377454Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.06621","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.357399Z","title":"Reward under attack: Analyzing the robustness and hackability of process reward models","venue":null,"work_id":"98c8e480-4613-4072-b7f8-621056dc4118","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:42348dc75b45b7979e4460fd28b6494350baec9c99a025f8f3496d618366cd35","observation_id":"f13d3bd3-8636-49bf-9f23-ca2825a91b1d","resolution":{"observed_at":"2026-07-03T21:18:58.359171Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.13602","last_updated":"2026-04-15T08:11:34Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T08:11:34Z","title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","version":1},"cited_work":{"arxiv_id":"2604.13602","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.13602","snapshot_observed_at":"2026-07-04T17:09:59.250632Z","title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","venue":"cs.LG","work_id":"fc71ddff-02d7-422e-852d-f4874cb845b5","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2604.13602","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:104ad5f9fdcee5b5ef326f206ae186ffc6bce42678fc0b034aa9371c3f0fe6f6","observation_id":"9aac31a8-6bc2-4a4a-a4b0-ac0388280c0b","resolution":{"observed_at":"2026-07-03T21:18:58.372138Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.15377","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.400107Z","title":"More test-time compute can hurt: Overestimation bias in llm beam search.arXiv preprint arXiv:2603.15377,","venue":null,"work_id":"484049da-c19b-45e1-a992-90138d650277","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:819681e1c871b9b44b82f18094a1956be4743b50effe8a2a5b47246a0196694b","observation_id":"fca526ff-4e66-4cca-8a5d-6cc9e535664d","resolution":{"observed_at":"2026-07-03T21:18:58.402092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.04648","last_updated":"2026-04-06T12:58:11Z","snapshot_observed_at":"2026-07-06T22:53:33.177713Z","submitted_at":"2026-04-06T12:58:11Z","title":"From Curiosity to Caution: Mitigating Reward Hacking for Best-of-N with Pessimism","version":1},"cited_work":{"arxiv_id":"2604.04648","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.04648","snapshot_observed_at":"2026-07-03T21:18:58.403370Z","title":"From Curiosity to Caution: Mitigating Reward Hacking for Best-of-N with Pessimism","venue":"cs.LG","work_id":"8f1b3a94-367a-41d7-9e78-0146025fcb8d","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2604.04648","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:0974158a229b9ce843f1fe79f1cabf89e7b5b2d3de28a893b8371739131195cb","observation_id":"4dd49caf-9621-40ee-ad3f-5fe8b710e869","resolution":{"observed_at":"2026-07-03T21:18:58.404824Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.19225","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T20:00:07.729442Z","title":"Junbo Li, Peng Zhou, Rui Meng, Meet P","venue":null,"work_id":"efacd88c-3686-4685-8f16-59a329bf18d6","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:9f69eed0f6824a37019e129417188f40255611e61cf073ab180d1396dec7288c","observation_id":"b1a72380-ad28-4828-af6e-72db290a8683","resolution":{"observed_at":"2026-07-03T21:18:58.333883Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.14209","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T14:27:07.367238Z","title":"Int: Self-proposed interventions enable credit assignment in llm reasoning","venue":null,"work_id":"32156c79-bbf0-4b3a-8319-d728be4814e8","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:5f10356d87421a186450c6cce682d2a6a76e748c8ee58b9ba994782de9317e99","observation_id":"47791efb-a617-49b5-bc38-87f1fdeec26e","resolution":{"observed_at":"2026-07-03T21:18:58.387573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.24022","last_updated":"2026-05-20T08:59:48Z","snapshot_observed_at":"2026-08-02T23:31:27.025065Z","submitted_at":"2026-05-20T08:59:48Z","title":"Adaptive KV Cache Reuse for Fast Long-Context LLM Serving","version":1},"cited_work":{"arxiv_id":"2605.24022","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.24022","snapshot_observed_at":"2026-07-03T21:18:58.406162Z","title":"Adaptive KV Cache Reuse for Fast Long-Context LLM Serving","venue":"cs.AR","work_id":"2d0dbf60-6653-40df-bda8-4ff074f5d826","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2605.24022","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:3a65107ec01116c9b73ccaaf2197974aeb6c39244cb09418a4c3200c6b0dbfd1","observation_id":"c7fe5dad-8229-4c64-84c9-62bd1b7988c1","resolution":{"observed_at":"2026-07-03T21:18:58.407427Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.22106","last_updated":"2026-05-21T07:40:57Z","snapshot_observed_at":"2026-07-06T23:32:30.578213Z","submitted_at":"2026-05-21T07:40:57Z","title":"ArborKV: Structure-Aware KV Cache Management for Scaling Tree-based LLM Reasoning","version":1},"cited_work":{"arxiv_id":"2605.22106","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.22106","snapshot_observed_at":"2026-07-03T21:08:58.855149Z","title":"ArborKV: Structure-Aware KV Cache Management for Scaling Tree-based LLM Reasoning","venue":"cs.AI","work_id":"ea00e12a-97da-4cdb-8f15-310c63c24484","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2605.22106","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:f41170414ff0fc0546b739408146cdf13b44f0e259d64bea9b793e339257f2f6","observation_id":"aa43bd65-83b7-400c-83d7-429dca86b6c5","resolution":{"observed_at":"2026-07-03T21:08:58.857588Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:710056d7c15827f674698a11d87ea191669b2e3b2f054950cfb32cf362ae3e84","observation_id":"3315d004-c551-43b2-9254-19e8a79e0799","resolution":{"observed_at":"2026-07-03T21:18:58.389583Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.21882","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.378211Z","title":"Closed-loop transformers: Autoregressive modeling as iterative latent equilibrium.arXiv preprint arXiv:2511.21882,","venue":null,"work_id":"d8025d30-5bf6-428e-8854-9f3f26bdc3ad","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:37625a134a597aeb42f7c704abaa53403d9714326878603d083f132dd4108720","observation_id":"b09bb816-eef6-443a-bba6-cee66ed2240f","resolution":{"observed_at":"2026-07-03T21:18:58.379863Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.07203","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.378732Z","title":"Mmrpt: Multimodal reinforcement pre-training via masked vision-dependent reasoning.arXiv preprint arXiv:2512.07203, 2025c","venue":null,"work_id":"7eb3166a-dd48-4ba0-83f5-7af3ca352df9","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:31a589ec648082d9c9426c044d8d94b7a4c0fce7f30f9b69df65a30786dcf04f","observation_id":"2369deb1-4c14-4f6b-8905-f0fb3c79a0bd","resolution":{"observed_at":"2026-07-03T21:18:58.380481Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.09065","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T01:09:18.612575Z","title":"arXiv preprint arXiv:2603.09065 , year=","venue":null,"work_id":"859390a3-8384-4b45-8eb8-3248be9e143f","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:ab3f5700fc45fae012e63debe108265ba032da692e39626fe043fb0793fc51c5","observation_id":"70e36fa2-6a44-4fcb-8375-76f2025b988d","resolution":{"observed_at":"2026-07-03T21:18:58.356277Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.25148","last_updated":"2026-06-18T03:33:32Z","snapshot_observed_at":"2026-08-04T13:45:31.993147Z","submitted_at":"2025-09-29T17:53:09Z","title":"AAPA: Adversarially Anchored Preference Alignment for Post-Training of Large Language Models","version":2},"cited_work":{"arxiv_id":"2509.25148","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.25148","snapshot_observed_at":"2026-07-03T21:18:58.373323Z","title":"AAPA: Adversarially Anchored Preference Alignment for Post-Training of Large Language Models","venue":"cs.AI","work_id":"6c0eead3-3aa2-4b5d-ad89-12c8f3faace0","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2509.25148","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:45c679d8242660c09555a0b3ee9a12144a27e13e9eeca6952d917096cd83f114","observation_id":"82e3a9ec-f72b-42da-b747-44ef9f62ab08","resolution":{"observed_at":"2026-07-03T21:18:58.374761Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.10567","last_updated":"2026-05-26T21:20:16Z","snapshot_observed_at":"2026-07-12T22:31:34.664599Z","submitted_at":"2026-04-12T10:26:41Z","title":"Early Decisions Matter: Proximity Bias and Initial Trajectory Shaping in Non-Autoregressive Diffusion Language Models","version":2},"cited_work":{"arxiv_id":"2604.10567","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.10567","snapshot_observed_at":"2026-07-03T21:18:58.393830Z","title":"Early Decisions Matter: Proximity Bias and Initial Trajectory Shaping in Non-Autoregressive Diffusion Language Models","venue":"cs.CL","work_id":"9f6d0660-e855-4910-b514-1a106641dc82","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2604.10567","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:94fe9eb17f1e8aa2580f6f0846955f8dd4008802607af942a44b0ecd4b2a2f94","observation_id":"4e5c98fd-1684-4b7a-b208-b492eb431894","resolution":{"observed_at":"2026-07-03T21:18:58.395252Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.08948","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.390754Z","title":"Corefine: Confidence-guided self-refinement for adaptive test-time compute.arXiv preprint arXiv:2602.08948, 2026a","venue":null,"work_id":"d8066d14-f5cd-4790-a665-f34f1a7fac20","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:9c20696eb763cf1d27ac5046a2ada60d831db68fd70a8c187b9b9993060e48a2","observation_id":"8ddeb62a-f527-43c7-b8ce-79a8a1b90ca1","resolution":{"observed_at":"2026-07-03T21:18:58.392569Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.18167","doi":"10.48550/arxiv.2506.18167","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding reasoning in thinking language models via steering vectors.arXiv preprint arXiv:2506.18167","venue":"arXiv (Cornell University)","work_id":"cf70c153-854b-4572-a6bb-96bcbbe896da","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:3779860d085e2b8351e15c72cca713fc985fce84fdc2b254276e1faa81ac76a0","observation_id":"fa684804-bab1-498f-a8d2-6b28a0d03fa7","resolution":{"observed_at":"2026-07-03T21:18:58.353528Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.19917","last_updated":"2026-04-14T07:00:09Z","snapshot_observed_at":"2026-07-06T22:43:12.468415Z","submitted_at":"2026-01-07T12:38:56Z","title":"PILOT: Planning via Internalized Latent Optimization Trajectories for Large Language Models","version":2},"cited_work":{"arxiv_id":"2601.19917","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.19917","snapshot_observed_at":"2026-07-03T21:18:58.380999Z","title":"PILOT: Planning via Internalized Latent Optimization Trajectories for Large Language Models","venue":"cs.CL","work_id":"d7e2a0ff-38ba-45b2-848f-9b7230932d12","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2601.19917","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:53cbe9f289da1262ce949e2b3b8e5650b95879218a9cef129ace08d472e28806","observation_id":"a84dbcc5-af56-4be8-99a1-7aa055112616","resolution":{"observed_at":"2026-07-03T21:18:58.382207Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.00977","last_updated":"2026-05-05T03:37:16Z","snapshot_observed_at":"2026-08-02T14:44:32.701926Z","submitted_at":"2026-03-01T08:09:03Z","title":"HiMAC: Hierarchical Macro-Micro Learning for Long-Horizon LLM Agents","version":2},"cited_work":{"arxiv_id":"2603.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.00977","snapshot_observed_at":"2026-07-03T21:18:58.336503Z","title":"HiMAC: Hierarchical Macro-Micro Learning for Long-Horizon LLM Agents","venue":"cs.AI","work_id":"1dad9110-9441-4ab5-92f1-8dfd47d6b79b","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2603.00977","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:58555d3355a2a77446405b26aa7628fd7afbc5c1b424badb92b9db29f60c12b6","observation_id":"00686bbf-8d21-4b3c-9757-0d3be16d8793","resolution":{"observed_at":"2026-07-03T21:18:58.338366Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T00:48:56.892634Z","title":"Probabilistic soundness guarantees in llm reasoning chains","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:d3af3bcd6929359c8a89be7429e18ed8c45baaf8f2aa6a065de58e6a191267db","observation_id":"fc57e249-3b1a-4fa1-828e-a015484f138d","resolution":{"observed_at":"2026-06-27T00:48:56.892634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.02479","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:18:58.358022Z","title":"Prism: Pushing the frontier of deep think via process reward model-guided inference.arXiv preprint arXiv:2603.02479,","venue":null,"work_id":"4d2d49af-1f5e-4a1c-937e-ccd43463bd63","year":null},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:024affa4feb864c5bc6981d7a21324b2ce8a20c8c15c7aa30970b03a4a4eaef0","observation_id":"89b0f538-d471-446b-ba6b-a47cfcb5424d","resolution":{"observed_at":"2026-07-03T21:18:58.359524Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.04304","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T12:53:50.245781Z","title":"v 1: Unifying generation and self-verification for parallel reasoners.arXiv preprint arXiv:2603.04304, 2026a","venue":null,"work_id":"7bbcc938-b7ac-4ba5-83f9-e8f0a7be8bb3","year":2026},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:9fd7abdd516637e8d0ee4de9647ecd85618844bf513bc33dcb9d3f894611890f","observation_id":"65c61e8d-c18c-4ffd-94d5-6f94025a9aec","resolution":{"observed_at":"2026-07-03T21:18:58.366082Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11456","last_updated":"2025-05-22T19:12:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-15T17:59:51Z","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","version":2},"cited_work":{"arxiv_id":"2504.11456","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.11456","snapshot_observed_at":"2026-07-09T03:25:57.840663Z","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","venue":"cs.CL","work_id":"3dea79f1-73da-4db2-8d7b-64013c3c5fa5","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2504.11456","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:7370e20db960dddeec35ce8a0ea625e01fb50e688c6846caaae00e731e391def","observation_id":"365dd4e4-44fb-4a6a-980e-69e8631d0689","resolution":{"observed_at":"2026-07-03T21:18:58.389882Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18071","last_updated":"2025-07-28T11:11:33Z","snapshot_observed_at":"2026-08-06T04:31:03.113409Z","submitted_at":"2025-07-24T03:50:32Z","title":"Group Sequence Policy Optimization","version":2},"cited_work":{"arxiv_id":"2507.18071","doi":"10.48550/arxiv.2507.18071","metadata_source":"pith","pith_arxiv_id":"2507.18071","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Group Sequence Policy Optimization","venue":"cs.LG","work_id":"3a98b53b-9f52-4d95-adf7-89353c0a9a65","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2507.18071","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:1629da9f73a983ab243b0c922fd3abaf0a861d3daea31bb4c30928b9f2d71224","observation_id":"10591c9e-6fd0-420f-a861-07e92f28ec32","resolution":{"observed_at":"2026-07-03T21:08:58.863995Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.12366","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:08:58.866914Z","title":"Disco: Re- inforcing large reasoning models with discriminative con- strained optimization.arXiv preprint arXiv:2505.12366","venue":null,"work_id":"195f0b28-c130-44e1-9b95-849040c53ea6","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:70923812577aca9b22a0ac7da87b412b32f5a6cf750a6081d93ae69fdfd1e6c7","observation_id":"8124e925-d29c-4936-a827-8f80984bd642","resolution":{"observed_at":"2026-07-03T21:08:58.868789Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T00:48:56.892634Z","title":"The proposed method does not involve human subjects, private user data, or personally identifiable information","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:6b0a803285321bc6a056fb48c7bc3257fe25b77c53b9e1683399b75736b630c9","observation_id":"24d57e2e-a177-4eea-86b4-6178b2760c23","resolution":{"observed_at":"2026-06-27T00:48:56.892634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9500.9750","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T21:08:58.870016Z","title":null,"venue":null,"work_id":"bb9636b2-3468-43c7-9099-624a7bfe9358","year":2023},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:58bd1cb23ea3f0cde71334445fae84f6532fb0d2045d948e9bcf7872c7cfe792","observation_id":"9c5c2945-2225-4fb8-a332-9faebfa4a26b","resolution":{"observed_at":"2026-07-03T21:08:58.871510Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":2,"verified_exact":32,"verified_fuzzy":0},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 0 inbound Pith citation observations for arXiv:2606.17735."}