{"as_of":"2026-08-21T23:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9a9a24cc42e10eeb0a50dcd2efe99a6c35307854dac612675b63cb19702199d8","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T00:53:30.969336Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.00301/citation-record","integrity":"/paper/2608.00301/integrity","json":"/paper/2608.00301/citation-record.json","paper":"/paper/2608.00301"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:25.817060Z","title":"and Zhang, Edwin , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:25.817060Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:d3f7003c2e4e05f0873b2dcf3b31a889e5c49315a3446f7d8c6819a2b17166f8","observation_id":"ab2705ea-ad33-47eb-825f-c2e3b31a29df","resolution":{"observed_at":"2026-08-04T00:53:25.817060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:25.888007Z","title":", year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:25.888007Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:26a4dde38932c5204034aa435779f7531460c2683b8eb56b3f738e70848ee9d1","observation_id":"2abb9b30-a985-4b00-943f-e9301274ae68","resolution":{"observed_at":"2026-08-04T00:53:25.888007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.16806","last_updated":"2026-05-15T00:00:51Z","snapshot_observed_at":"2026-08-07T06:17:41.140558Z","submitted_at":"2025-07-22T17:56:01Z","title":"Beyond Binary Rewards: Training LMs to Reason About Their Uncertainty","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.16806","snapshot_observed_at":"2026-08-04T00:53:26.000469Z","title":"Beyond Binary Rewards: Training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.000469Z"},"links":{"cited_paper":"/paper/2507.16806","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:7d76cfbb4b38ba15098db05c38de3dfd68acba0b0421c69c01cea6e8568c4f07","observation_id":"57826263-223c-4742-be08-037d3faa7ccf","resolution":{"observed_at":"2026-08-04T00:53:26.000469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.25760","last_updated":"2026-06-08T21:01:17Z","snapshot_observed_at":"2026-08-19T22:19:52.572059Z","submitted_at":"2025-09-30T04:25:17Z","title":"TruthRL: Incentivizing Truthful LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.25760","snapshot_observed_at":"2026-08-04T00:53:26.113437Z","title":"2509.25760 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.113437Z"},"links":{"cited_paper":"/paper/2509.25760","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:072d1f360837a11dd53175638f75765427b7e96293b71043f10c63f2a756a729","observation_id":"676c5148-e63d-4158-bbf2-c2f8b97722f6","resolution":{"observed_at":"2026-08-04T00:53:26.113437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.25850","last_updated":"2026-05-25T13:42:37Z","snapshot_observed_at":"2026-08-17T03:53:46.092851Z","submitted_at":"2026-05-25T13:42:37Z","title":"TIAR: Trajectory-Informed Advantage Reweighting for LLM Abstention Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.25850","snapshot_observed_at":"2026-08-04T00:53:26.206601Z","title":"2605.25850 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.206601Z"},"links":{"cited_paper":"/paper/2605.25850","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:ab59f4e6b0765ed70002bef202849a84e6209bee47e87285c61cd6cca56f9d8c","observation_id":"1d0ec2be-5923-4e84-8d32-c4160d977077","resolution":{"observed_at":"2026-08-04T00:53:26.206601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.22779","last_updated":"2026-04-03T11:30:56Z","snapshot_observed_at":"2026-08-21T06:56:52.119883Z","submitted_at":"2026-04-03T11:30:56Z","title":"KARL: Mitigating Hallucinations in LLMs via Knowledge-Boundary-Aware Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.22779","snapshot_observed_at":"2026-08-04T00:53:26.307965Z","title":"2604.22779 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.307965Z"},"links":{"cited_paper":"/paper/2604.22779","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:949ce1464c29256b983e4c9d73bfc976a541c4f6fa16d5f19cbf31a7a6aaa2ca","observation_id":"54305e50-0fb7-46ae-be6e-c870cb08f6e0","resolution":{"observed_at":"2026-08-04T00:53:26.307965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.22648","last_updated":"2026-05-26T11:07:20Z","snapshot_observed_at":"2026-08-16T07:38:14.972424Z","submitted_at":"2026-01-30T07:07:42Z","title":"UCPO: Uncertainty-Aware Policy Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.22648","snapshot_observed_at":"2026-08-04T00:53:26.423440Z","title":"2601.22648 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.423440Z"},"links":{"cited_paper":"/paper/2601.22648","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:6752761e1bda9dbbcf202cf930c949b055ee02e0f8e3f12e5b6a40f12e75885f","observation_id":"04ea5d40-19af-43f6-9220-e1a19046f1ea","resolution":{"observed_at":"2026-08-04T00:53:26.423440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.17073","last_updated":"2026-04-18T17:21:40Z","snapshot_observed_at":"2026-08-15T21:25:50.291903Z","submitted_at":"2026-04-18T17:21:40Z","title":"Abstain-R1: Calibrated Abstention and Post-Refusal Clarification via Verifiable RL","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.17073","snapshot_observed_at":"2026-08-04T00:53:26.556359Z","title":"Abstain-","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.556359Z"},"links":{"cited_paper":"/paper/2604.17073","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:6ab73d70ff51eaea5929c728039679f6e9cc1d5d62841770df0502508342ffd0","observation_id":"fef8b9ba-3f0d-43a2-adfa-7c7b669dd7bb","resolution":{"observed_at":"2026-08-04T00:53:26.556359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:26.707407Z","title":"Enhancing Reliability across Short and Long-Form","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.707407Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:47b93be7f241ef2385f7d7cd3ae0cff332653ca9567170896d9a7e3d611af47e","observation_id":"3f7d78d5-7b3a-4d41-9312-00c744b3dfb1","resolution":{"observed_at":"2026-08-04T00:53:26.707407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18349","last_updated":"2024-08-08T08:57:23Z","snapshot_observed_at":"2026-08-16T14:05:59.232389Z","submitted_at":"2024-03-27T08:39:56Z","title":"Rejection Improves Reliability: Training LLMs to Refuse Unknown Questions Using RL from Knowledge Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18349","snapshot_observed_at":"2026-08-04T00:53:26.819594Z","title":"Rejection Improves Reliability: Training","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.819594Z"},"links":{"cited_paper":"/paper/2403.18349","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:09efb8e9fccd383fe8401cc12aa2d7a27c0cce6d6f808e22f9f00d89dd55f48b","observation_id":"74247269-072c-4a8b-a849-8dbaae796dad","resolution":{"observed_at":"2026-08-04T00:53:26.819594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:26.884202Z","title":"The Hallucination Tax of Reinforcement Finetuning , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.884202Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:cde525131f38d4d3b4e32d34ee793a66d2ff4245c324d0baa52fe913bb5d2d02","observation_id":"fdec0010-df81-43b6-9070-93d785f61194","resolution":{"observed_at":"2026-08-04T00:53:26.884202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:26.972189Z","title":"2509.17730 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:26.972189Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:6d5bdd84ce5a2f17b8d24705b8b89d9d73197c5e92c17b26163bd67498267c99","observation_id":"d5138077-ffb5-40e6-9fb9-6dc1749a82c3","resolution":{"observed_at":"2026-08-04T00:53:26.972189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:27.061604Z","title":"Knowledge-Level Consistency Reinforcement Learning: Dual-Fact Alignment for Long-Form Factuality , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.061604Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:daf25ea00892347b5c0975b377dc8d4e3f102a980dbbf36e9f5620beb4eb9c98","observation_id":"ef39490a-612b-4557-a6b6-ad575cf526ed","resolution":{"observed_at":"2026-08-04T00:53:27.061604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:27.195630Z","title":"2505.13529 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.195630Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:5f71786d0c3fb6c16080acfce2ea8405802502f03f47ea9a5734fa10c1c9c17f","observation_id":"c3f7dae9-8e02-4c1d-b9b8-aee43e2b6828","resolution":{"observed_at":"2026-08-04T00:53:27.195630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:27.338083Z","title":"Vanishing Gradients in Reinforcement Finetuning of Language Models , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.338083Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:58ce61dc6cc4ba5e109c439585755308115be1856606216b10f7a04345e3c74f","observation_id":"9ca9606d-b173-4b24-a852-b638efab5d88","resolution":{"observed_at":"2026-08-04T00:53:27.338083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.06392","last_updated":"2022-06-02T06:16:17Z","snapshot_observed_at":"2026-08-15T19:04:43.031317Z","submitted_at":"2020-05-13T16:01:39Z","title":"On the Global Convergence Rates of Softmax Policy Gradient Methods","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.06392","snapshot_observed_at":"2026-08-04T00:53:27.440871Z","title":"2020 , title =","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.440871Z"},"links":{"cited_paper":"/paper/2005.06392","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:c50e2a7ce009b14f853f03a39830e291ea38ee6030a5d74d033c331946ab3b76","observation_id":"48d7edbf-492b-4f17-ad60-91b93f9125f3","resolution":{"observed_at":"2026-08-04T00:53:27.440871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:27.529142Z","title":"The Entropy Mechanism of Reinforcement Learning for Reasoning Language Models , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.529142Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:966146fc3f53ed298d0bcb8467397912b81a772965ed047e0fbd97cf8945b58a","observation_id":"c4689bc2-5a92-442e-bde5-03b62c4c55e5","resolution":{"observed_at":"2026-08-04T00:53:27.529142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-08-14T15:51:49.712613Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-04T00:53:27.653093Z","title":"Understanding the Effects of","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.653093Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:ef12450921122d7136c0de8fceafb1e6bdddff82522b133f0d660292dc36a968","observation_id":"05b2a897-3b9e-419b-aadc-8f88a138c232","resolution":{"observed_at":"2026-08-04T00:53:27.653093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1901.11275","last_updated":"2019-06-04T07:44:24Z","snapshot_observed_at":"2026-08-14T17:23:11.083967Z","submitted_at":"2019-01-31T09:10:08Z","title":"A Theory of Regularized Markov Decision Processes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1901.11275","snapshot_observed_at":"2026-08-04T00:53:27.851497Z","title":"A Theory of Regularized","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.851497Z"},"links":{"cited_paper":"/paper/1901.11275","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:4dec4266a11328c7fb6f8ba1da5ceb48c433cd4862b6139d8270e0dc5fce1604","observation_id":"2b1af7e4-cad4-4e02-8915-1aa5cb3720c6","resolution":{"observed_at":"2026-08-04T00:53:27.851497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2003.14089","last_updated":"2021-01-06T14:12:57Z","snapshot_observed_at":"2026-08-05T04:41:53.389161Z","submitted_at":"2020-03-31T10:55:06Z","title":"Leverage the Average: an Analysis of KL Regularization in RL","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.14089","snapshot_observed_at":"2026-08-04T00:53:27.989253Z","title":"Leverage the Average: an Analysis of","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.989253Z"},"links":{"cited_paper":"/paper/2003.14089","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:145c2ce73884d637480007faaca0e413dced3754ad9c97735cce7c511e4bf140","observation_id":"352f964e-166d-4f4f-8e61-4f50a47f5763","resolution":{"observed_at":"2026-08-04T00:53:27.989253Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-04T00:53:28.131447Z","title":"2402.03300 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.131447Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:34c2d37b98ace0433f8e7a99f18a4d38dc3fa1b270bbffdeffa765d94dbfecf5","observation_id":"5cb1418d-f3ba-4e48-abb8-6c1abf01582c","resolution":{"observed_at":"2026-08-04T00:53:28.131447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-08-13T12:34:54.476684Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-04T00:53:28.214601Z","title":"Understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.214601Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:4d7f671d53aff5c9b8f24e558e5cff8d372e6a56264cbea3b600ddef2968b876","observation_id":"34d3b5f7-0649-4dcc-af3c-139723f97f59","resolution":{"observed_at":"2026-08-04T00:53:28.214601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-08-09T14:30:33.899591Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-08-04T00:53:28.299798Z","title":"2024 , title =","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.299798Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:7ffad208d923c74c1836aff6a60ba53a981cbde9a6873dc2672ee669eeae38f2","observation_id":"ab733218-cc76-46b6-9b54-775408158617","resolution":{"observed_at":"2026-08-04T00:53:28.299798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-18T05:01:20.543826Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-04T00:53:28.428593Z","title":"2503.14476 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.428593Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:272913741b580e318876f873726732553190692c4d0099bd7edeb2465a4cceff","observation_id":"b4773189-261c-489d-8eb2-2ad401d27a4e","resolution":{"observed_at":"2026-08-04T00:53:28.428593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03262","last_updated":"2025-11-10T15:11:13Z","snapshot_observed_at":"2026-08-16T04:57:30.418363Z","submitted_at":"2025-01-04T02:08:06Z","title":"REINFORCE++: Stabilizing Critic-Free Policy Optimization with Global Advantage Normalization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03262","snapshot_observed_at":"2026-08-04T00:53:28.530409Z","title":"2501.03262 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.530409Z"},"links":{"cited_paper":"/paper/2501.03262","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:51bf5d206bd3266321ffbc929d577035533eed73ec4d5acc0b9f4bdc089bb361","observation_id":"2f5698f8-f765-4fbe-a238-c6f77a46ffbd","resolution":{"observed_at":"2026-08-04T00:53:28.530409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:28.668683Z","title":"Scaling Laws for Reward Model Overoptimization , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.668683Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:85e05fc908d74459e53a5dd338206767e51474bd66c9f3e45068ea79b3f7bcfb","observation_id":"4a2fb1da-8723-4123-9238-55217ffb70fc","resolution":{"observed_at":"2026-08-04T00:53:28.668683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:28.847058Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.847058Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:3e54a379fdbefdf6ff67f747789e97ccb1548c6ed09e6659574b6fb9486db0ae","observation_id":"f2005d49-85be-4b61-a798-a3125ee9fea5","resolution":{"observed_at":"2026-08-04T00:53:28.847058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:28.926687Z","title":"and Martic, Miljan and others , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.926687Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:7a79eb254ffe5495198a2d8f42ac75bcabc5b4346eb450e39298c667c20861fa","observation_id":"5cea912c-8fb8-426c-9511-ada9ed84a4ae","resolution":{"observed_at":"2026-08-04T00:53:28.926687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:28.981204Z","title":"and others , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:28.981204Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:291ecf423194ac55e6d91d74bc96484238cf589605d7d09edacc18669831de37","observation_id":"da372744-d219-44ed-9d85-ee50eaf7f8d6","resolution":{"observed_at":"2026-08-04T00:53:28.981204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.084946Z","title":"Training Language Models to Follow Instructions with Human Feedback , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.084946Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:7d419aef7d8fd90ff84991bde225b3afc5c47bd0e1225cf51f201adda1550638","observation_id":"6a66bdae-0c77-466d-93b0-ff8b2785630e","resolution":{"observed_at":"2026-08-04T00:53:29.084946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.148434Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.148434Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:5909cfb0f761551c433fc39b095cda2488a7c7f36c72a2c86bdd48120d63d8f8","observation_id":"94218d1a-a73b-44d1-9a4f-e0c056e33543","resolution":{"observed_at":"2026-08-04T00:53:29.148434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.203476Z","title":"On the Foundations of Noise-free Selective Classification , journal =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.203476Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:e79d2452b0023ed6c1b89d9aa9d26d41287bd77554375847b43639d47e96f001","observation_id":"d9ebbbce-2630-4dc0-83ea-267f841fa17e","resolution":{"observed_at":"2026-08-04T00:53:29.203476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.208690Z","title":"Selective Classification for Deep Neural Networks , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.208690Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:9102887ce188e6c84e71d1ba3d9f8697588011de2732461f6c7732e653cf139c","observation_id":"312536c2-1da6-43f1-b6f6-0a97df50dd84","resolution":{"observed_at":"2026-08-04T00:53:29.208690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.275043Z","title":"Selective Question Answering under Domain Shift , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.275043Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:2cf71c56555aa8f8d8d897a5b25de47471a6138f14c5cd8a8e9e7f9ab7a01a15","observation_id":"b6d3ec95-38a8-4e46-92a4-1d6507a674c2","resolution":{"observed_at":"2026-08-04T00:53:29.275043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.487503Z","title":"Out-of-Distribution Detection and Selective Generation for Conditional Language Models , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.487503Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:e9f80dcc97717a9da93a2638995594c1eceb21e5533c901058bdbc50011a39c2","observation_id":"84154b10-a95b-4b4a-b71b-8a7515f75373","resolution":{"observed_at":"2026-08-04T00:53:29.487503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.603289Z","title":"Language Models (Mostly) Know What They Know , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.603289Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:3db9d7c9a3e6a976a5166025cebf7c5159ec51ce455217a43a89522c5fa835fd","observation_id":"82234b30-2e58-45fa-8300-7f711880624d","resolution":{"observed_at":"2026-08-04T00:53:29.603289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.696163Z","title":"Teaching Models to Express Their Uncertainty in Words , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.696163Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:1bee7e0a154e0273adb66340a090ba67aa928a18e6bfeab2bbd88ed665709f2f","observation_id":"5fb6dfa4-3519-424a-b12a-94c35fdf8497","resolution":{"observed_at":"2026-08-04T00:53:29.696163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:29.836202Z","title":"Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.836202Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:fb375364f63eca68425cc1f612efb65a3361f28e005197e16ce3b00a524c4704","observation_id":"3793a836-01b9-40e1-b150-26d2d3d9b44a","resolution":{"observed_at":"2026-08-04T00:53:29.836202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.29490","last_updated":"2026-06-28T16:39:18Z","snapshot_observed_at":"2026-08-13T04:34:05.057189Z","submitted_at":"2026-06-28T16:39:18Z","title":"Reported Confidence in LLMs Tracks Commitment More Than Correctness","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.29490","snapshot_observed_at":"2026-08-04T00:53:29.955373Z","title":"Reported Confidence in","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:29.955373Z"},"links":{"cited_paper":"/paper/2606.29490","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:ee64f94ed0e694afb499d8daf678e2218e7a874e3e4c794b7c956eb529ee1383","observation_id":"6bb6a822-fbde-45a1-920e-d56aa29c8c99","resolution":{"observed_at":"2026-08-04T00:53:29.955373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.086006Z","title":", year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.086006Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:0455e353120fdc0e9f429fd160b81a448e604f7903f4c71aa44b813c93c5aa0e","observation_id":"2f7a097b-94e0-402c-8c92-2de6209a4dc2","resolution":{"observed_at":"2026-08-04T00:53:30.086006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.187212Z","title":"Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.187212Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:5f0ac3336c3774ec3cf276609c07b6935a9c7fd65f8b5600611a9f46f38f86f9","observation_id":"ae44adc7-85fd-42af-afc3-f4f6f889dd0e","resolution":{"observed_at":"2026-08-04T00:53:30.187212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01563","last_updated":"2024-04-04T11:32:03Z","snapshot_observed_at":"2026-08-16T14:03:40.226273Z","submitted_at":"2024-04-04T11:32:03Z","title":"Mitigating LLM Hallucinations via Conformal Abstention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.01563","snapshot_observed_at":"2026-08-04T00:53:30.253798Z","title":"Mitigating","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.253798Z"},"links":{"cited_paper":"/paper/2405.01563","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:40dddc1703d03df7a22dbe8ddb8948aec04c8c4e0c84b50b55faec5ad1baf5f4","observation_id":"e3e7f2be-5139-4f2a-b3f4-b3c0afd1ee4c","resolution":{"observed_at":"2026-08-04T00:53:30.253798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-04T00:53:30.337364Z","title":"2412.15115 , archivePrefix =","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.337364Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:63ab2e5c417beb4ce42e57bad4173f3118830c6a92de63c4e566c0d775f6fb93","observation_id":"dc84d3f3-b641-4ecf-89fd-0dea84786c69","resolution":{"observed_at":"2026-08-04T00:53:30.337364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1705.03551","last_updated":"2017-05-13T21:12:37Z","snapshot_observed_at":"2026-08-13T08:58:47.096400Z","submitted_at":"2017-05-09T21:35:07Z","title":"TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.03551","snapshot_observed_at":"2026-08-04T00:53:30.434651Z","title":"and Zettlemoyer, Luke , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.434651Z"},"links":{"cited_paper":"/paper/1705.03551","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:66ebc765e59013494c2f38247c0c82df305a5e909aac55932132347156e356a0","observation_id":"894f0cf3-5f82-4d9f-8219-47967dfb7db4","resolution":{"observed_at":"2026-08-04T00:53:30.434651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.529942Z","title":"When Not to Trust Language Models: Investigating Effectiveness of Parametric and Non-Parametric Memories , eprint =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.529942Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:17397b7a9a83b4eccec7872238aeecc56bf4a1186a7b6018414d67836a916980","observation_id":"9a01facf-b339-482a-9fbf-418f418d2f16","resolution":{"observed_at":"2026-08-04T00:53:30.529942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.589609Z","title":"2511.13029 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.589609Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:4af82bacedef850f785fb48897c53c38bfa6b75a1e0f8b4f43cd07a7bf66e160","observation_id":"24ce766d-cdd2-4133-991c-c12bf47fb26d","resolution":{"observed_at":"2026-08-04T00:53:30.589609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.694160Z","title":", year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.694160Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:2a3efedecec20a36274863d993e6cbf238351641fbfaa6664df0549821cdb746","observation_id":"c5a2ff5f-f9dc-425f-acd1-3dc3e29d3f8c","resolution":{"observed_at":"2026-08-04T00:53:30.694160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.769359Z","title":", year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.769359Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:61f7a397e8042be0d717630cde8cf66c107aee369d876d520cd2054fe1e60c21","observation_id":"264e1248-4b57-4457-b282-25a819bc00a1","resolution":{"observed_at":"2026-08-04T00:53:30.769359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.816895Z","title":", year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.816895Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:d0edbabe5f433be46df892ab813ae7a558180705ad482368ef833c1caac36d75","observation_id":"8f3000b0-2145-4245-b08d-d9a984197b8a","resolution":{"observed_at":"2026-08-04T00:53:30.816895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.876999Z","title":", year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.876999Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:575760d108bfd3960ae32c5d82d793c6a4f064c7ba4d7b012b37b2827b02dd84","observation_id":"d3efdc51-f862-4439-8739-f1cf7d1215df","resolution":{"observed_at":"2026-08-04T00:53:30.876999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:53:30.969336Z","title":"Approximating","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:30.969336Z"},"links":{"citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:f30a85fff8455f54dba8328e7fd0d5fe8ac9cb8b7acc44253ab35092e4fd7275","observation_id":"681405dc-2711-4b67-a25b-d529ac63e90d","resolution":{"observed_at":"2026-08-04T00:53:30.969336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T19:42:48.000510Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":51,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 0 inbound Pith citation observations for arXiv:2608.00301."}