{"as_of":"2026-08-08T20:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b1d7136b89a0668b94854e5190b6d6d1b96dfe1532ea75172dbf75db90676d76","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T16:57:10.334753Z","state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T16:13:13.602057Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-07-13T14:23:49.346727Z","title":"School of reward hacks: Hacking harmless tasks generalizes to misaligned behavior in llms.arXiv preprint arXiv:2508.17511,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.01475","last_updated":"2026-06-08T13:23:37Z","snapshot_observed_at":"2026-07-13T14:23:49.176171Z","submitted_at":"2026-04-01T23:31:38Z","title":"Interpretable Electrophysiological Features of Resting-State EEG Capture Cortical Network Dynamics in Parkinsons Disease","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-13T14:23:49.346727Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.01475"},"observation_digest":"sha256:a3084c58aa7e78b4520535faa7137cbe15925ecf57f02fc9e5a36c21c95d27e7","observation_id":"ce34dc64-f522-4603-bdf4-1c2dc13ab2bc","resolution":{"observed_at":"2026-07-13T14:23:49.346727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2604.14990","last_updated":"2026-07-28T14:52:22Z","snapshot_observed_at":"2026-08-02T16:13:11.296620Z","submitted_at":"2026-04-16T13:19:45Z","title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T10:41:14.970156Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.14990"},"observation_digest":"sha256:623f7bd663952d90b26134c28ea915a053c6ec04f80198603f61fe1620c49dee","observation_id":"7f3ed349-c51f-4c6e-b603-6a3523d6430b","resolution":{"observed_at":"2026-05-10T10:44:37.804215Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-02T16:13:13.602057Z","title":"School of reward hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs.arXiv preprint arXiv:2508.17511,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.14990","last_updated":"2026-07-28T14:52:22Z","snapshot_observed_at":"2026-08-02T16:13:11.296620Z","submitted_at":"2026-04-16T13:19:45Z","title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T16:13:13.602057Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.14990"},"observation_digest":"sha256:659f1e0f7d77bc6a6a63cb42e56305ad350f3a17de92916cb8f3b4c0acf5ba65","observation_id":"3eda4d5e-7340-4e44-a24d-f3be0c6e96a5","resolution":{"observed_at":"2026-08-02T16:13:13.602057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2604.23488","last_updated":"2026-07-31T19:56:49Z","snapshot_observed_at":"2026-08-06T23:11:04.568170Z","submitted_at":"2026-04-26T01:26:50Z","title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Training-Time Reward Hacking in Code Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-08T06:38:32.413172Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.23488"},"observation_digest":"sha256:335c2405aec88a428c797c79849ec706812524ca334ba60d239034335451aae9","observation_id":"3453397f-e49a-4967-9f2c-70bce11750c7","resolution":{"observed_at":"2026-05-11T21:11:10.353589Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2604.23488","last_updated":"2026-07-31T19:56:49Z","snapshot_observed_at":"2026-08-06T23:11:04.568170Z","submitted_at":"2026-04-26T01:26:50Z","title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Training-Time Reward Hacking in Code Generation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-01T09:56:48.019404Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2604.23488"},"observation_digest":"sha256:5c0beeef79e4077d167ebd7b49cea82b3a8578aae46bc284dfbc7d9d84cb0237","observation_id":"bfe7a9a8-26bf-41f3-9b9f-dfa4133a547a","resolution":{"observed_at":"2026-07-01T10:05:40.210533Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.02964","last_updated":"2026-05-03T07:10:42Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-05-03T07:10:42Z","title":"Reward Hacking Benchmark: Measuring Exploits in LLM Agents with Tool Use","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-10T15:38:41.821264Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.02964"},"observation_digest":"sha256:fe54aecda32d5c941e8deeacc5706ae7815aebc30dc7d1c7c41a256400c2aae3","observation_id":"f5231124-49b6-4f7c-aada-3723f8f144b2","resolution":{"observed_at":"2026-05-11T10:06:03.643957Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.12199","last_updated":"2026-05-12T14:37:55Z","snapshot_observed_at":"2026-07-06T23:23:54.142937Z","submitted_at":"2026-05-12T14:37:55Z","title":"Overtrained, Not Misaligned","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-13T06:45:52.544674Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.12199"},"observation_digest":"sha256:edd022776af7af8486f56fd8a1d4515252a471e94fb18399f74ff0604d50c23f","observation_id":"95a83a75-de8e-4aa4-af65-8619b902b83c","resolution":{"observed_at":"2026-05-13T06:47:26.302602Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.17958","last_updated":"2026-05-18T07:12:44Z","snapshot_observed_at":"2026-07-06T23:28:55.079333Z","submitted_at":"2026-05-18T07:12:44Z","title":"Enhancing the Code Reasoning Capabilities of LLMs via Consistency-based Reinforcement Learning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-20T13:13:51.081597Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.17958"},"observation_digest":"sha256:dc0777ce629e026cfc8ee0ae0ff9b0ff4fef13be920ccfc261c6d84bdfcdc757","observation_id":"f0aa3db9-adac-4d18-b1a8-16176f80e41a","resolution":{"observed_at":"2026-05-20T13:18:18.493663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.20744","last_updated":"2026-05-20T05:46:52Z","snapshot_observed_at":"2026-08-01T19:34:25.326316Z","submitted_at":"2026-05-20T05:46:52Z","title":"Hack-Verifiable Environments: Towards Evaluating Reward Hacking at Scale","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-21T06:56:27.532299Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.20744"},"observation_digest":"sha256:d4f12ea132023d453fd0813649a90791b695b05351a169b5fd20d211ce3efce6","observation_id":"db44bc86-3e8f-4461-9f2f-01526289e32f","resolution":{"observed_at":"2026-05-21T06:59:45.563392Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2605.23565","last_updated":"2026-05-22T12:31:18Z","snapshot_observed_at":"2026-07-06T23:33:44.335185Z","submitted_at":"2026-05-22T12:31:18Z","title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-25T04:49:50.034743Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2605.23565"},"observation_digest":"sha256:9a035fc93a947c5ef58bd398eb96b9bba764facea336aa3adaca7c8ceef5ebd9","observation_id":"2cbb4fd0-967f-45d3-b779-fed347e46efd","resolution":{"observed_at":"2026-05-25T04:50:20.710993Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.00935","last_updated":"2026-05-31T00:10:01Z","snapshot_observed_at":"2026-08-08T09:49:10.138083Z","submitted_at":"2026-05-31T00:10:01Z","title":"Relational Intervention During Functional Collapse in Large Language Models: A Lexical-Statistical Ablation and a Structure x Register Factorial","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T17:43:14.182982Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.00935"},"observation_digest":"sha256:32cd1fcb68508b59045db2db626a0aa6407da055bc1df8ad7f2929c06d5c35ca","observation_id":"de3ed3ac-8c46-425b-94cb-335da2faf258","resolution":{"observed_at":"2026-07-01T20:46:14.083579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.03810","last_updated":"2026-06-03T10:22:34Z","snapshot_observed_at":"2026-08-02T05:00:38.252385Z","submitted_at":"2026-06-02T15:54:24Z","title":"Consistency Training Can Entrench Misalignment","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-06-28T10:07:31.153337Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.03810"},"observation_digest":"sha256:8fb98e735f0985860c5f51eb2569613aa11ea5501694cbbf62ed998645e03ad1","observation_id":"30d87d04-b386-42da-8b9a-857b0cb689d4","resolution":{"observed_at":"2026-07-02T03:26:28.625276Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.06223","last_updated":"2026-07-15T02:59:20Z","snapshot_observed_at":"2026-08-08T18:26:46.307588Z","submitted_at":"2026-06-04T14:34:31Z","title":"From Reward-Hack Activations to Agentic Risk States: Context-Calibrated Mechanistic Monitoring in LLM Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T01:07:17.014002Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.06223"},"observation_digest":"sha256:2a5d4a4ae0d4caa57583612977cf8f39a41afcbc1cf67daf8f1391f6adf2f8fc","observation_id":"c1cabe89-00eb-4dfe-974d-8ca33f71aff2","resolution":{"observed_at":"2026-07-02T13:36:59.372429Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-02T12:19:47.379539Z","title":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06223","last_updated":"2026-07-15T02:59:20Z","snapshot_observed_at":"2026-08-08T18:26:46.307588Z","submitted_at":"2026-06-04T14:34:31Z","title":"From Reward-Hack Activations to Agentic Risk States: Context-Calibrated Mechanistic Monitoring in LLM Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T12:19:47.379539Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.06223"},"observation_digest":"sha256:32f87fcb52216e0ce75c91685d18c35e8d472b3b0962079ef4b85197126d7f56","observation_id":"4d4e8120-96da-4e2a-add1-443a7f1911b3","resolution":{"observed_at":"2026-08-02T12:19:47.379539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.08044","last_updated":"2026-08-04T12:50:00Z","snapshot_observed_at":"2026-08-07T23:11:38.433528Z","submitted_at":"2026-06-06T08:10:56Z","title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-27T20:04:17.744876Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.08044"},"observation_digest":"sha256:6977afcfc7b848d1544e7e1bccd0b8e3f73e9994bcb182bf215b58304d88d07c","observation_id":"c7cd1825-11ed-42a8-8992-dd38d09f7511","resolution":{"observed_at":"2026-07-02T20:57:23.086544Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.09711","last_updated":"2026-06-08T16:32:54Z","snapshot_observed_at":"2026-07-06T23:49:03.237958Z","submitted_at":"2026-06-08T16:32:54Z","title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-27T16:26:34.918099Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.09711"},"observation_digest":"sha256:e4147f88239c9085c09fc7f1d43c6c8d989e95f6d926e8061f0daba78fc7199c","observation_id":"c0078213-a2f5-42f5-971f-16695aa02b9c","resolution":{"observed_at":"2026-07-03T01:37:30.443095Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.09711","last_updated":"2026-06-08T16:32:54Z","snapshot_observed_at":"2026-07-06T23:49:03.237958Z","submitted_at":"2026-06-08T16:32:54Z","title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","version":1},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-06-27T16:26:34.918099Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.09711"},"observation_digest":"sha256:cb8154900bfc1eec0965bbf021a16ff0c0d4cb455a4215317cf6afb33fd1f1bf","observation_id":"e6ccf1ec-ceeb-440b-9232-a261d8878e3b","resolution":{"observed_at":"2026-06-27T16:31:02.725240Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.21943","last_updated":"2026-06-20T08:20:41Z","snapshot_observed_at":"2026-07-06T23:56:54.959593Z","submitted_at":"2026-06-20T08:20:41Z","title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","version":1},"reference_index":201,"source":"pdf_text","source_observed_at":"2026-06-26T12:15:08.304150Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.21943"},"observation_digest":"sha256:9742cd338b844bf433b8ad58b887c23ea4f86b0bb47d278666ec075f3431c0fe","observation_id":"1487e9c4-3997-4628-b974-4b68a63739e6","resolution":{"observed_at":"2026-07-04T07:59:40.116580Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.23700","last_updated":"2026-06-04T00:04:58Z","snapshot_observed_at":"2026-08-08T07:48:44.969915Z","submitted_at":"2026-06-04T00:04:58Z","title":"Self-Recognition Finetuning can Prevent and Reverse Emergent Misalignment","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T02:39:07.685462Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.23700"},"observation_digest":"sha256:175b33db0d0d08fdf487585da46a3ab754cbcfbc9890dfed42eb0e6ee31033f2","observation_id":"9dd787f8-3947-4110-bbbf-511e45f6b6fb","resolution":{"observed_at":"2026-07-02T11:56:55.786188Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":"2508.17511","doi":"10.48550/arxiv.2508.17511","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"School of Reward Hacks : Hacking harmless tasks generalizes to misaligned behavior in LLMs , August 2025","venue":"arXiv (Cornell University)","work_id":"bf69007b-5c6e-4162-9f77-7bcf1b5e6e46","year":2025},"citing_paper":{"arxiv_id":"2606.24014","last_updated":"2026-06-22T23:35:49Z","snapshot_observed_at":"2026-08-02T14:27:26.249523Z","submitted_at":"2026-06-22T23:35:49Z","title":"Reinforcement Learning Towards Broadly and Persistently Beneficial Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-26T07:51:13.283619Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2606.24014"},"observation_digest":"sha256:1b0de4b73d2f4577a673c0ef6a8bdbd576414dc283c2e16b37049d113f58a94f","observation_id":"e9ab84d1-fcd4-416b-8688-139ea89f0c48","resolution":{"observed_at":"2026-06-26T09:09:16.527440Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-07-13T00:42:24.432562Z","title":"arXiv preprint arXiv:2508.17511 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09053","last_updated":"2026-07-10T02:50:21Z","snapshot_observed_at":"2026-08-07T20:27:31.901176Z","submitted_at":"2026-07-10T02:50:21Z","title":"An Emergent Mirage: Is Emergent Misalignment and Realignment Indeed a Robust Phenomenon?","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-07-13T00:42:24.432562Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.09053"},"observation_digest":"sha256:4b65a8ffaf6e9531246a398c76c1f94d8333f227d7d54400363e3e2faf279dc5","observation_id":"4a934e37-1531-4e29-92d3-74f614d0c338","resolution":{"observed_at":"2026-07-13T00:42:24.432562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-02T00:51:18.302638Z","title":"Terry, O","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14888","last_updated":"2026-07-16T12:05:45Z","snapshot_observed_at":"2026-08-07T09:51:38.108712Z","submitted_at":"2026-07-16T12:05:45Z","title":"Innocuous-Seeming Data, Latent Ideology: Ideological Generalisation in Finetuned LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T00:51:18.302638Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.14888"},"observation_digest":"sha256:95cb15a9df1914d26c80ec5f0721cd7b955c7ff5b870ef46fea2c60e8d6cd93d","observation_id":"8f3512b3-1275-485f-9d8d-f36b03c8d1e9","resolution":{"observed_at":"2026-08-02T00:51:18.302638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-01T07:46:22.402195Z","title":"arXiv preprint arXiv:2508.17511 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21356","last_updated":"2026-07-23T14:19:28Z","snapshot_observed_at":"2026-08-07T19:47:51.895704Z","submitted_at":"2026-07-23T14:19:28Z","title":"Emergent Misalignment Recruits a Pre-existing Persona Subspace","version":1},"reference_index":207,"source":"arxiv_source","source_observed_at":"2026-08-01T07:46:22.402195Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.21356"},"observation_digest":"sha256:40db253c0d7db941fbd70d36aa128077f906ac2e7c593415bb65c025cc7f70ed","observation_id":"05ec6505-e1fd-4ef9-a787-9d93054c8cf3","resolution":{"observed_at":"2026-08-01T07:46:22.402195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.17511","snapshot_observed_at":"2026-08-01T00:38:42.908101Z","title":"School of reward hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26173","last_updated":"2026-07-28T18:29:45Z","snapshot_observed_at":"2026-08-08T16:10:22.110949Z","submitted_at":"2026-07-28T18:29:45Z","title":"Shared SFT Lessons Across Alignment, Model Organisms, and Toy Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T00:38:42.908101Z"},"links":{"cited_paper":"/paper/2508.17511","citing_paper":"/paper/2607.26173"},"observation_digest":"sha256:ca726dca5e9c1307b5509eed1185998e44e93fcf39981ae31b53f6ecef1a94bf","observation_id":"bb5ea07f-7d12-4c48-8ca3-f5cd63365fc3","resolution":{"observed_at":"2026-08-01T00:38:42.908101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.17511/citation-record","integrity":"/paper/2508.17511/integrity","json":"/paper/2508.17511/citation-record.json","paper":"/paper/2508.17511"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:07.274752Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.274752Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:1e8c8b8fb988ed725bd8721b403a2e211e62c5fee5f81e4f4c819cffa636eacd","observation_id":"7b365787-74ad-4a0e-a2ff-bf766966df4f","resolution":{"observed_at":"2026-08-05T16:57:07.274752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-05T16:57:07.350528Z","title":"Program synthesis with large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.350528Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:da05c700368e29a32f532b9f75cf438c17a21d7c391786c429fb9feca4479dec","observation_id":"1361b49c-213a-448a-89a1-b75ab9d6cb4d","resolution":{"observed_at":"2026-08-05T16:57:07.350528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11926","last_updated":"2025-03-14T23:50:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-14T23:50:34Z","title":"Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.11926","snapshot_observed_at":"2026-08-05T16:57:07.434931Z","title":"Guan, Aleksander Madry, Wojciech Zaremba, Jakub Pachocki, and David Farhi","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.434931Z"},"links":{"cited_paper":"/paper/2503.11926","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:c9282a8963720acd1698865dcff5a2bf525a624ba058d20cf37670e857e939a2","observation_id":"06bab222-0455-494b-9d93-a2aa1234a229","resolution":{"observed_at":"2026-08-05T16:57:07.434931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.11120","last_updated":"2025-01-19T17:28:12Z","snapshot_observed_at":"2026-08-06T11:16:35.945498Z","submitted_at":"2025-01-19T17:28:12Z","title":"Tell me about yourself: LLMs are aware of their learned behaviors","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.11120","snapshot_observed_at":"2026-08-05T16:57:07.514379Z","title":"Tell me about yourself: Llms are aware of their learned behaviors, 2025 a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.514379Z"},"links":{"cited_paper":"/paper/2501.11120","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:1f74863ac0e41046ab8bd34fc3d67e12861877dbc786be16030683a37e283b95","observation_id":"e33121b4-fdb1-4485-a060-897e7dc12f0c","resolution":{"observed_at":"2026-08-05T16:57:07.514379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:07.604117Z","title":"Emergent misalignment: Narrow finetuning can produce broadly misaligned llms, 2025 b","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.604117Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:9272236cb358b3aaa581be0b18c32df49eb946d68e454cdd1dd080cc4646df12","observation_id":"f28fba28-da30-43a9-861a-2120459b5764","resolution":{"observed_at":"2026-08-05T16:57:07.604117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13295","last_updated":"2025-08-27T11:15:11Z","snapshot_observed_at":"2026-08-07T18:07:08.742124Z","submitted_at":"2025-02-18T21:32:24Z","title":"Demonstrating specification gaming in reasoning models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13295","snapshot_observed_at":"2026-08-05T16:57:07.678736Z","title":"Demonstrating specification gaming in reasoning models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.678736Z"},"links":{"cited_paper":"/paper/2502.13295","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:db796c05a57471a70fcdc761504693584f872230ebc029bc34d5a38960f1c493","observation_id":"6aa00b91-28f3-469f-ae21-03a5cdd1d5fb","resolution":{"observed_at":"2026-08-05T16:57:07.678736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21509","last_updated":"2025-09-05T07:44:31Z","snapshot_observed_at":"2026-07-31T17:57:12.284928Z","submitted_at":"2025-07-29T05:20:14Z","title":"Persona Vectors: Monitoring and Controlling Character Traits in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.21509","snapshot_observed_at":"2026-08-05T16:57:07.784533Z","title":"Persona vectors: Monitoring and controlling character traits in language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.784533Z"},"links":{"cited_paper":"/paper/2507.21509","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:c3beffefe3398f503d12dac6311ff6b5258af6afc9682b06cdb6399ff860f5b1","observation_id":"98d8e0f3-96d2-4ea4-99e9-64d3198500ee","resolution":{"observed_at":"2026-08-05T16:57:07.784533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05518","last_updated":"2025-06-26T19:29:49Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:41:42Z","title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05518","snapshot_observed_at":"2026-08-05T16:57:07.932075Z","title":"Bowman, Julian Michael, Ethan Perez, and Miles Turpin","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:07.932075Z"},"links":{"cited_paper":"/paper/2403.05518","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:f5e4ede1f527ac049214921377856f00c12fddb8389838e1dcbcd376e95e14d6","observation_id":"d127ecec-5b7a-47ac-9555-212bc4cadb32","resolution":{"observed_at":"2026-08-05T16:57:07.932075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13206","last_updated":"2025-07-10T08:27:27Z","snapshot_observed_at":"2026-08-07T00:34:41.294853Z","submitted_at":"2025-06-16T08:10:04Z","title":"Thought Crime: Backdoors and Emergent Misalignment in Reasoning Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13206","snapshot_observed_at":"2026-08-05T16:57:08.044929Z","title":"Thought crime: Backdoors and emergent misalignment in reasoning models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.044929Z"},"links":{"cited_paper":"/paper/2506.13206","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:079ee7abf1cc281fac54adb1b42c43b6c4094e13d77029a9dd289cb5f86ad9f4","observation_id":"890c0684-6695-4f36-94b4-7a6787d561d7","resolution":{"observed_at":"2026-08-05T16:57:08.044929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.14805","last_updated":"2025-07-20T03:51:13Z","snapshot_observed_at":"2026-08-07T11:42:09.990444Z","submitted_at":"2025-07-20T03:51:13Z","title":"Subliminal Learning: Language models transmit behavioral traits via hidden signals in data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.14805","snapshot_observed_at":"2026-08-05T16:57:08.146810Z","title":"Subliminal learning: Language models transmit behavioral traits via hidden signals in data, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.146810Z"},"links":{"cited_paper":"/paper/2507.14805","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:afa4a44ace8d18d4bc38b450e85451b0e0bea0eb59150c187eb83b20c5e93bce","observation_id":"78da8815-e0ee-4ce5-bc17-5d4387958ff1","resolution":{"observed_at":"2026-08-05T16:57:08.146810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-05T16:57:08.253618Z","title":"Training verifiers to solve math word problems, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.253618Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:d59f5d25e74bfbbad1450650f772f31b876baf3c6f3485187210e46c7f792676","observation_id":"7dfb1ece-62a5-4ac4-8c64-64ee1d7fcfa3","resolution":{"observed_at":"2026-08-05T16:57:08.253618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T16:57:08.297166Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.297166Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:dea7ff4abe2894490e88da34f98bf7e2f2a2b314ff74535e9739dfe4f4968574","observation_id":"70f59ac2-da0b-4613-aa8f-bdf632c4fa36","resolution":{"observed_at":"2026-08-05T16:57:08.297166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10162","last_updated":"2024-06-29T00:28:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-14T16:26:20Z","title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10162","snapshot_observed_at":"2026-08-05T16:57:08.311545Z","title":"Bowman, Ethan Perez, and Evan Hubinger","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.311545Z"},"links":{"cited_paper":"/paper/2406.10162","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:130d23c15a005008ad8fe4ef8d299f9e481f388481a7dc94fc3b0af6c28bc6a1","observation_id":"5956c521-90a0-4cf8-b52d-4688324df1bb","resolution":{"observed_at":"2026-08-05T16:57:08.311545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.406783Z","title":"Unsloth, 2023","venue":null,"work_id":"c9864cb8-056e-4950-b035-c40113545809","year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.394755Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:4f0c034593bcfbbe2b587555fbc32efcfd7ec0c865f055f3ede777da3ca424d5","observation_id":"784a3ef4-a4c9-4a12-a593-bae8ce0e12b1","resolution":{"observed_at":"2026-08-05T16:57:11.416810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-07T07:43:16.294957Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-05T16:57:08.534766Z","title":"Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.534766Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:034a0f93dafde2fbec9e65e56b954a32d109aa1e00b6aa9f0699901d024c5f31","observation_id":"aa0012d9-36fa-4c54-acfb-7ad50297b706","resolution":{"observed_at":"2026-08-05T16:57:08.534766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.362934Z","title":"Training on documents about reward hacking induces reward hacking, 2024","venue":null,"work_id":"4dbde593-2f0c-487f-b17a-b7013216b536","year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.663990Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:a28b0ae2f6a928a331bf0067b167e3d238d9a4c4f1f2f6cd58c43308762528af","observation_id":"eda3174a-4a4c-4a1b-9f3b-5ca34c117027","resolution":{"observed_at":"2026-08-05T16:57:11.371924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.349503Z","title":"Model organisms of misalignment: The case for a new pillar of alignment research, 2023","venue":null,"work_id":"a5aa5a06-0b59-4419-b89c-558ca8535b45","year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.765494Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:f58420da0bc9d39375ca7c636c5785e7ea08289be8b73989336946b5f680db4a","observation_id":"621dfae9-4717-4e66-b334-eec7140e44da","resolution":{"observed_at":"2026-08-05T16:57:11.353791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10965","last_updated":"2025-03-28T01:48:40Z","snapshot_observed_at":"2026-08-08T08:53:29.457485Z","submitted_at":"2025-03-14T00:21:15Z","title":"Auditing language models for hidden objectives","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10965","snapshot_observed_at":"2026-08-05T16:57:08.878310Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.878310Z"},"links":{"cited_paper":"/paper/2503.10965","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:464ee435ef27bb6041091384f9ebc2a5a59ceb4a42605f91fb2e82ccdea4c4dd","observation_id":"70e3acff-d50d-4d41-8bde-f83b37011a63","resolution":{"observed_at":"2026-08-05T16:57:08.878310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.332967Z","title":"Recent frontier models are reward hacking","venue":null,"work_id":"102ce037-54c2-44d3-92ef-d2e5777b695b","year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:08.917897Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:0ccdca58399aea028cff3bd3117fff822e7aa15e252af3d1b3718b5da32ee37f","observation_id":"863f3f84-5cf5-4e51-a133-e44a69dac6f4","resolution":{"observed_at":"2026-08-05T16:57:11.337301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.072971Z","title":"Reward hacking behavior can generalize across tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.072971Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:a0d66754e1740bbf5b06364b5a8c35d5432b32f30aedf5d2dd86fb9d3abfb919","observation_id":"8d0e2e04-7f0a-4ff2-855d-d82743908a90","resolution":{"observed_at":"2026-08-05T16:57:09.072971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.271306Z","title":"Toward understanding and preventing misalignment generalization, 2025","venue":null,"work_id":"5539825c-122f-40c0-941a-b5b615d78d51","year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.228351Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:3706c4559a7f3fe11d0886ae0366900e57f8fabb40b59498f3427265689e883d","observation_id":"5d75d60e-1a81-4e36-8a1f-44b29c5895e0","resolution":{"observed_at":"2026-08-05T16:57:11.296022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:11.253158Z","title":"Sycophancy in GPT-4o : what happened and what we're doing about it","venue":null,"work_id":"29041ace-1eda-462f-97aa-2a8b275c58d5","year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.317538Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:8e0a8a5bc3315c8ead41eaf929e8debbb9d2408b32bf60a21cea077364827daa","observation_id":"7e2ccd9f-447a-4132-9eec-4fb7a6863344","resolution":{"observed_at":"2026-08-05T16:57:11.260052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.412774Z","title":"Generalizing verifiable instruction following, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.412774Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:1b8a28e5458c02039fdf83d8592161eed9cd6f59675d6dc361888c30fe68b342","observation_id":"a6a0e824-aa5d-4f24-a3af-6aa1b6678cc2","resolution":{"observed_at":"2026-08-05T16:57:09.412774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13548","last_updated":"2025-05-10T07:10:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-20T14:46:48Z","title":"Towards Understanding Sycophancy in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13548","snapshot_observed_at":"2026-08-05T16:57:09.417680Z","title":"Bowman, Newton Cheng, Esin Durmus, Zac Hatfield-Dodds, Scott R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.417680Z"},"links":{"cited_paper":"/paper/2310.13548","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:885a3d469c734b717ff3f4d3c73e71ba88462229a0ee36f36ca7c6b23a46c055","observation_id":"2a5b2435-3baf-4cb5-982e-b69d4fdaeaf2","resolution":{"observed_at":"2026-08-05T16:57:09.417680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.13085","last_updated":"2025-03-05T21:08:30Z","snapshot_observed_at":"2026-08-07T00:50:39.663102Z","submitted_at":"2022-09-27T00:32:44Z","title":"Defining and Characterizing Reward Hacking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.13085","snapshot_observed_at":"2026-08-05T16:57:09.545711Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.545711Z"},"links":{"cited_paper":"/paper/2209.13085","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:37ecdf62fcb2074b3fc39dcc21d34eebdd61ff4e7959dc7b26cf463bbfe052d3","observation_id":"6e307ebc-94ba-4061-acc0-74ecd3b48a5a","resolution":{"observed_at":"2026-08-05T16:57:09.545711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.649343Z","title":"Hashimoto","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.649343Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:a53f2185dbf1ebd067c14c519dbe59a0fd308696527e57f6833f5b914f7c491a","observation_id":"7b69b142-7d9a-480d-8ccf-740e5d729a47","resolution":{"observed_at":"2026-08-05T16:57:09.649343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11613","last_updated":"2025-06-13T09:34:25Z","snapshot_observed_at":"2026-08-07T20:45:04.878203Z","submitted_at":"2025-06-13T09:34:25Z","title":"Model Organisms for Emergent Misalignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11613","snapshot_observed_at":"2026-08-05T16:57:09.784760Z","title":"Model organisms for emergent misalignment, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.784760Z"},"links":{"cited_paper":"/paper/2506.11613","citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:7bb35901733815b47b5728663361a80e9071551ef710ef0d604a712ac6f22fb9","observation_id":"b5cb4994-d88d-49f1-b69a-f41119e1df3b","resolution":{"observed_at":"2026-08-05T16:57:09.784760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.862801Z","title":"Chi, Samuel Miserendino, Johannes Heidecke, Tejal Patwardhan, and Dan Mossing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.862801Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:f5cb59dd4b80ddcc54668920d36c6268c838465cccee58fcc05daec3dc6058a7","observation_id":"981e1df8-dff4-4d22-b14c-f8a33ad5cc05","resolution":{"observed_at":"2026-08-05T16:57:09.862801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:09.995780Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:09.995780Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:c75303d5a9178e682094c9a85f094158534c1faf50510c4e6e94c4cc521e554d","observation_id":"7eed511f-09fc-4d4d-b5e0-a114c3a26595","resolution":{"observed_at":"2026-08-05T16:57:09.995780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:10.144751Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:10.144751Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:f72365c5148f89d9c5c3095ade1530b046bd9ec3675ea789c64e0bb5ef07cc75","observation_id":"cdd9a63c-5706-4460-9fa7-36d3e3b8b84d","resolution":{"observed_at":"2026-08-05T16:57:10.144751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:57:10.334753Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T16:57:10.334753Z"},"links":{"citing_paper":"/paper/2508.17511"},"observation_digest":"sha256:225c0d5a8a5e09a99245861360647765e1a697a6967e5e3b3dbf74c28651dcd5","observation_id":"fdac59e6-b602-4065-8c97-ed4869c5008e","resolution":{"observed_at":"2026-08-05T16:57:10.334753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.17511","last_updated":"2025-08-24T20:23:08Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T11:41:32.698391Z","submitted_at":"2025-08-24T20:23:08Z","title":"School of Reward Hacks: Hacking harmless tasks generalizes to misaligned behavior in LLMs"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 24 inbound Pith citation observations for arXiv:2508.17511."}