{"as_of":"2026-08-18T11:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e8a80f91675e4e8a5615d4aaccc715d716b08fb46b99c8d07ac19716162de566","coverage":[{"denominator":77,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":77,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T21:11:02.149478Z","state":"measured"},{"denominator":80,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":80,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:51:56.450815Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-17T00:31:24.634056Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.10670","snapshot_observed_at":"2026-08-07T04:51:56.450815Z","title":"Interpretable risk mitigation in llm agent systems.arXiv preprint arXiv:2505.10670, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09420","last_updated":"2025-06-11T06:08:13Z","snapshot_observed_at":"2026-08-09T00:55:40.500387Z","submitted_at":"2025-06-11T06:08:13Z","title":"A Call for Collaborative Intelligence: Why Human-Agent Systems Should Precede AI Autonomy","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:51:56.450815Z"},"links":{"cited_paper":"/paper/2505.10670","citing_paper":"/paper/2506.09420"},"observation_digest":"sha256:feeac579c59424f523aec23174c09e48fbb6f8f53d206fcaf90c78c388b85d38","observation_id":"b72a9458-11dc-47d1-84f0-f77d9122b3d6","resolution":{"observed_at":"2026-08-07T04:51:56.450815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"cited_work":{"arxiv_id":"2505.10670","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.10670","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","venue":null,"work_id":"d2817880-ea6c-45a7-9d7d-8469adabd169","year":2025},"citing_paper":{"arxiv_id":"2512.05929","last_updated":"2026-06-29T18:18:24Z","snapshot_observed_at":"2026-08-16T07:11:14.391624Z","submitted_at":"2025-12-05T18:12:21Z","title":"LLM Harms: A Taxonomy and Discussion","version":2},"reference_index":189,"source":"pdf_text","source_observed_at":"2026-05-17T00:29:07.951709Z"},"links":{"cited_paper":"/paper/2505.10670","citing_paper":"/paper/2512.05929"},"observation_digest":"sha256:13f33b4cc959f8832a5b13787cd58932f80352ef1252bb541e1a483df69a18ec","observation_id":"1b7081ac-e99d-43fc-be2e-190be39eed65","resolution":{"observed_at":"2026-05-17T00:31:24.636618Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.10670","snapshot_observed_at":"2026-08-03T18:19:27.337242Z","title":"Interpretable Risk Mitigation in LLM Agent Systems,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.05929","last_updated":"2026-06-29T18:18:24Z","snapshot_observed_at":"2026-08-16T07:11:14.391624Z","submitted_at":"2025-12-05T18:12:21Z","title":"LLM Harms: A Taxonomy and Discussion","version":4},"reference_index":189,"source":"pdf_text","source_observed_at":"2026-08-03T18:19:27.337242Z"},"links":{"cited_paper":"/paper/2505.10670","citing_paper":"/paper/2512.05929"},"observation_digest":"sha256:8accc8b8ee2bb53ffb54ba41fb6b8e0c96d7edee279df5fdd9a7697468d0a490","observation_id":"84d058c8-51a1-4859-970e-8b0c97117aac","resolution":{"observed_at":"2026-08-03T18:19:27.337242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.10670/citation-record","integrity":"/paper/2505.10670/integrity","json":"/paper/2505.10670/citation-record.json","paper":"/paper/2505.10670"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.638268Z","title":"Artificial intelligence and the future of work: Evidence from OECD countries","venue":null,"work_id":"4e93b0ea-6d02-427d-82c6-c8da5df0ea01","year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.782535Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:99522b61a1151fbea3f396245014bcad51b26299b9fa2536e84ebb170c9560c9","observation_id":"c43eafac-f59d-4479-8e8e-f89c0279a822","resolution":{"observed_at":"2026-08-15T21:11:03.643143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.622368Z","title":"Dai, Chelsea Finn, Justin Fu, Kanishka Gopalakrishnan, et al","venue":null,"work_id":"4cfae7f4-561d-41bc-a0d9-7f211940f04a","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.787366Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:2f25a7efecc3fa8451a62347e7ac9d94d083d3a4cd4fe544bc65ed079a46e2bd","observation_id":"bf3d3a32-c723-4ab7-a21a-243b73e9489b","resolution":{"observed_at":"2026-08-15T21:11:03.627827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.607866Z","title":"Mistral 7b: Open foundation models, 2023","venue":null,"work_id":"d52424b5-fcf8-4fd8-900b-9ce56746b04f","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.792100Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:839a9aa20eca3f6daee89d14baff0e31ebbe5063a98239c477fdc16441eb2264","observation_id":"3801ed77-8037-4a96-b87d-1b592e49f25f","resolution":{"observed_at":"2026-08-15T21:11:03.612474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16867","last_updated":"2025-05-07T12:44:45Z","snapshot_observed_at":"2026-08-18T11:09:48.072216Z","submitted_at":"2023-05-26T12:17:59Z","title":"Playing repeated games with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16867","snapshot_observed_at":"2026-08-15T21:11:01.796811Z","title":"Playing repeated games with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.796811Z"},"links":{"cited_paper":"/paper/2305.16867","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:6ca29cfc59c340584f5b203cd765e5bce4de7d393f10b82246de7654a43706db","observation_id":"54ed5ba8-5bf9-4e48-a28b-7b79fb7685e8","resolution":{"observed_at":"2026-08-15T21:11:01.796811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.06565","last_updated":"2016-07-25T17:23:29Z","snapshot_observed_at":"2026-07-06T05:00:46.434335Z","submitted_at":"2016-06-21T13:37:05Z","title":"Concrete Problems in AI Safety","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.06565","snapshot_observed_at":"2026-08-15T21:11:01.802560Z","title":"Concrete problems in AI safety","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.802560Z"},"links":{"cited_paper":"/paper/1606.06565","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:5cd9982d7d255a715d4fbd06e77d7bd27c056690ac283d7827bdf9f04b35441a","observation_id":"9b7eff9b-d167-419d-ae41-da6ca8c46fe7","resolution":{"observed_at":"2026-08-15T21:11:01.802560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.592741Z","title":null,"venue":null,"work_id":"794a2695-9d64-47cf-9316-2d7aeafb5ec7","year":1984},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.807791Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:b394d1ddae1402c491577ee1f5aaa6ec8742d63d534cda666a61d49125ad0bde","observation_id":"16702f8c-2cb7-4d44-a2c2-8387f4b0ae83","resolution":{"observed_at":"2026-08-15T21:11:03.597680Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-15T21:11:01.813065Z","title":"Training a helpful and harmless assistant with reinforcement learning from human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.813065Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:cfb3e2020694388452c6130215ee5ea7a7bf8bebc8067527262df8508c99f766","observation_id":"94b9346b-084f-4af5-a678-a12f9bd12757","resolution":{"observed_at":"2026-08-15T21:11:01.813065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.575085Z","title":"Emergent tool use from multi-agent autocurricula","venue":null,"work_id":"dd733df4-15f3-4a2b-9ae8-50f4e56b74e0","year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.817697Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:681ae3f19be051cc2f4746212c2fd0d625beea03eb5ea31294178b5f4f49f041","observation_id":"5e6eafe6-a930-4240-a93e-aaa30f5aa5d6","resolution":{"observed_at":"2026-08-15T21:11:03.580794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.556424Z","title":"Bender, Timnit Gebru, Angelina McMillan-Major, and Shmargaret Shmitchell","venue":null,"work_id":"7eebd975-9ca2-4abc-8f1e-d2e40b0611f0","year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.822090Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:f16df6b11ccba79916cf3ab4ad847861b0cf88ec26ac6c78c9a8dd4dbf2e734e","observation_id":"351ad7f1-6ab1-403d-b7ff-ef072b69c117","resolution":{"observed_at":"2026-08-15T21:11:03.562891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-15T21:11:01.826932Z","title":"Hudson, Ehsan Adeli, Russ Altman, Simran Arora, Sydney von Arx, et al","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.826932Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:f66ba15a4101c183ed099580e49199bbe9914502a8aae97ce54ff0d0921aa369","observation_id":"8a106779-c66d-48c7-8c48-b0f78d8abf43","resolution":{"observed_at":"2026-08-15T21:11:01.826932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.538368Z","title":null,"venue":null,"work_id":"cdf4c402-e4e3-47b5-92a9-7832bfd81d54","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.832003Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:9df3b0bb6f64b0d8da056ce734905f456e0a0efa9d5c5c695945a4f1e542474e","observation_id":"2a7adc57-b623-480d-be0c-b1dd7e69a6ee","resolution":{"observed_at":"2026-08-15T21:11:03.544642Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-15T21:11:01.836214Z","title":"RT-1: Robotics transformer for real-world control at scale","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.836214Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:37ee356306d056c24f581471d08ee04fefc707811c081a99ae379af6d87e718d","observation_id":"a338b78c-533d-4e4b-9b30-639f80356236","resolution":{"observed_at":"2026-08-15T21:11:01.836214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.521438Z","title":"Playing games with gpt: What can we learn about a large language model from canonical strategic games? SSRN Electronic Journal, 2023","venue":null,"work_id":"daec497b-2379-4a00-acba-e406c24b6c01","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.840725Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:a91742897ea7b9d76c9a3c4139248b954120b9adfec8a425a2e13687723016c8","observation_id":"16d22cc2-e43f-4ec2-aff9-61a6d878a1a6","resolution":{"observed_at":"2026-08-15T21:11:03.526152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.503212Z","title":"Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D","venue":null,"work_id":"2ddb56ad-1b11-4050-a6e4-1cf601cd82dd","year":1901},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.844793Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:0850297141c211f40a5c303048de21e0aa6ca8f6d357c43a7f191c7b802089ef","observation_id":"fa8b734c-b330-4c95-85f8-426bab50f0df","resolution":{"observed_at":"2026-08-15T21:11:03.508895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.485636Z","title":"What can machine learning do? workforce implications","venue":null,"work_id":"e4531483-9fe0-4eb4-9335-4f27f6b3199e","year":2017},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.848928Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:2a4bb93ef97d411d8c53048d29f52bcd2233b169453d765cf7078d41394b0d4b","observation_id":"b60e728a-df0f-4060-9aeb-3d455d436d74","resolution":{"observed_at":"2026-08-15T21:11:03.491497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-15T21:11:01.853101Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.853101Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:97ed16a42de506f8718cd4f08902971027df67c63e7a6780ed23beab556a3633","observation_id":"e55a86e7-747f-4031-8209-67c0bab512ab","resolution":{"observed_at":"2026-08-15T21:11:01.853101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.857296Z","title":"Instigating cooperation among llm agents using adaptive information modulation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.857296Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:4264b9f34f01334f2678e1fc3514d949b85cd5a1279239a2593ca68902214fd3","observation_id":"08b4f22d-3581-40e0-aa27-dd14aaa9041a","resolution":{"observed_at":"2026-08-15T21:11:01.857296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08600","last_updated":"2023-10-04T13:17:38Z","snapshot_observed_at":"2026-08-16T13:37:26.527260Z","submitted_at":"2023-09-15T17:56:55Z","title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08600","snapshot_observed_at":"2026-08-15T21:11:01.870732Z","title":"Sparse autoencoders find highly interpretable features in language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.870732Z"},"links":{"cited_paper":"/paper/2309.08600","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e7c442b67768c15dfba60d5b0138870ffbd84b34ce604d5a4e567c45bccc3048","observation_id":"72433d46-a595-481b-a5ef-2f80a6405e69","resolution":{"observed_at":"2026-08-15T21:11:01.870732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.453516Z","title":"Reinforcement learning in a prisoner’s dilemma","venue":null,"work_id":"c52095a1-54e2-45ff-97a0-fa0936d831fd","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.875550Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e535ecd7bd48b6aee31bda7f12a1d6abff10c35a4b8f9e94fa2bc86266ec38b2","observation_id":"eadd38c1-fbcd-4b04-a105-2ea6f1e6a5e6","resolution":{"observed_at":"2026-08-15T21:11:03.458432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.10652","last_updated":"2022-09-21T20:49:26Z","snapshot_observed_at":"2026-08-16T21:36:28.067615Z","submitted_at":"2022-09-21T20:49:26Z","title":"Toy Models of Superposition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.10652","snapshot_observed_at":"2026-08-15T21:11:01.880133Z","title":"Toy models of superposition","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.880133Z"},"links":{"cited_paper":"/paper/2209.10652","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:0992c45d236b9be6b8100dfca13e7f9e6bf7606178c95e13fec6818d1f48fe87","observation_id":"24f6a202-78a8-481d-8db8-1c4b79df229d","resolution":{"observed_at":"2026-08-15T21:11:01.880133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.04866","last_updated":"2022-12-19T17:54:22Z","snapshot_observed_at":"2026-08-16T16:25:31.937056Z","submitted_at":"2022-10-10T17:34:49Z","title":"PoGaIN: Poisson-Gaussian Image Noise Modeling from Paired Samples","version":2},"cited_work":{"arxiv_id":"2210.04866","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.04866","snapshot_observed_at":"2026-08-15T21:11:02.828852Z","title":"PoGaIN: Poisson-Gaussian Image Noise Modeling from Paired Samples","venue":"cs.CV","work_id":"fa2a04ea-1db8-4f70-933c-c9bce17001f1","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.886176Z"},"links":{"cited_paper":"/paper/2210.04866","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:d513a1956e2b86de4a014cf35c788237d4a49c58f95160d4c1a1b332755c8ebd","observation_id":"3da5c266-6d9b-45d6-8d90-16eafac12480","resolution":{"observed_at":"2026-08-15T21:11:02.834041Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14860","last_updated":"2025-02-27T03:03:59Z","snapshot_observed_at":"2026-08-18T08:28:54.917862Z","submitted_at":"2024-05-23T17:59:04Z","title":"Not All Language Model Features Are One-Dimensionally Linear","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14860","snapshot_observed_at":"2026-08-15T21:11:01.891128Z","title":"Michaud, Wes Gurnee, and Max Tegmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.891128Z"},"links":{"cited_paper":"/paper/2405.14860","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:a4a89d07ed85bac39988892460f20381dbca7d5d9de495807f84b63a47275e8a","observation_id":"9a93bcfe-f4cc-4a52-9e56-a716c5208000","resolution":{"observed_at":"2026-08-15T21:11:01.891128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.438321Z","title":"Some experimental games","venue":null,"work_id":"25e408a7-0f48-4cb6-bfaa-49df36c93acb","year":1958},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.896121Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:49a0005a56c8bc9b5eacee8682f06590a5622667ad5f883445c94e2f50e60969","observation_id":"80ab46e3-96c8-4341-b15b-4ef484f62fc1","resolution":{"observed_at":"2026-08-15T21:11:03.442507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13605","last_updated":"2024-09-19T15:19:58Z","snapshot_observed_at":"2026-08-16T13:41:25.329780Z","submitted_at":"2024-06-19T14:51:14Z","title":"Nicer Than Humans: How do Large Language Models Behave in the Prisoner's Dilemma?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13605","snapshot_observed_at":"2026-08-15T21:11:01.900891Z","title":"Nicer than humans: How do large language models behave in the prisoner’s dilemma? ArXiv, abs/2406.13605, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.900891Z"},"links":{"cited_paper":"/paper/2406.13605","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:6830411567da8c1bf1f9781f3a722c9651e98ff91090dcffd4850d363b390e4b","observation_id":"f9da16f4-a983-41b4-b6d2-cccdb6ba0154","resolution":{"observed_at":"2026-08-15T21:11:01.900891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.906112Z","title":"Artificial intelligence, values, and alignment","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.906112Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:96b276e3f4c71d4a78f4c570cd540c156fb9c5cbfe859d019e78c71da890881a","observation_id":"88637004-c8d5-4677-890d-d755e3357d38","resolution":{"observed_at":"2026-08-15T21:11:01.906112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T21:11:01.910539Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.910539Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:c7f71b82858392a2a34dd0bb5095bac27db7c730f2fc7ba922ce24037474570f","observation_id":"cb688a1f-ae7d-4316-9fcd-3ddffe2c900b","resolution":{"observed_at":"2026-08-15T21:11:01.910539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-15T21:11:01.915548Z","title":"Measuring massive multitask language understanding, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.915548Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:d2694fc08a587683ada08213828ad884ae52aed2f21b2a8e09bc4c36d2978238","observation_id":"2aea34d7-c236-4ba2-b05c-5266d05cbe3a","resolution":{"observed_at":"2026-08-15T21:11:01.915548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.920013Z","title":"Measuring mathematical problem solving with the math dataset,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.920013Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:9de668d2f9727cb07dd65a9ba0119ee301e1b5fb8d295eef3b6cbd7bee0ccbd9","observation_id":"daa69867-514e-4500-ac68-9b455cec0cf3","resolution":{"observed_at":"2026-08-15T21:11:01.920013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.411699Z","title":"Cogagent: A visual language model for gui agents","venue":null,"work_id":"ef5b7c60-0d6e-49bb-867b-689d2084ff61","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.930482Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:ca928a69b51caf319ca122ec93ca755a3b8eedbfb364c840e3ab3348a8622dd3","observation_id":"ccc257f3-639b-4bd7-b693-d99e2d88ec53","resolution":{"observed_at":"2026-08-15T21:11:03.416984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.395348Z","title":"Non-linear inference time intervention: Improving llm truthfulness","venue":null,"work_id":"64b1abf9-c196-413a-85b6-c036384036ef","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.935781Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:844c2ba95bb5f8284019d11d65b584ff648afcab7dfe3efdd991937db16f7939","observation_id":"f465e08f-8581-42e6-b51b-25e319bc463c","resolution":{"observed_at":"2026-08-15T21:11:03.400826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10403","last_updated":"2023-05-26T17:59:33Z","snapshot_observed_at":"2026-08-06T19:23:30.079167Z","submitted_at":"2022-12-20T16:29:03Z","title":"Towards Reasoning in Large Language Models: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10403","snapshot_observed_at":"2026-08-15T21:11:01.939936Z","title":"Towards reasoning in large language models: A survey, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.939936Z"},"links":{"cited_paper":"/paper/2212.10403","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:da374a1ea748d380dd9343439240bd041a27de20ed7e00495e38cba27908d737","observation_id":"79c959d9-a3b3-4327-86ed-96acb55e005b","resolution":{"observed_at":"2026-08-15T21:11:01.939936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.379909Z","title":"Large language models for uavs: Current state and pathways to the future","venue":null,"work_id":"7c2885cd-da7b-49bf-b8b4-ea6021e14e35","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.944592Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:581457afea64933ece197f20a12b724c7232bf75ecdd605a113a910a8be2c642","observation_id":"f8bf0ff6-e450-4f1c-b07c-950516506ed9","resolution":{"observed_at":"2026-08-15T21:11:03.384589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-13T19:43:49.936776Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-15T21:11:01.949339Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.949339Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:870cfd79364ac20bacd101a3d80de9627d50789c0df5f68bcedfa90b96ffc25c","observation_id":"a7a39f6f-26fb-4c28-b196-bc2a237200ff","resolution":{"observed_at":"2026-08-15T21:11:01.949339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.360951Z","title":"llama-3-8b-it-res (revision 53425c3), 2024","venue":null,"work_id":"3a878928-d5e4-46ed-a09f-8b9fbf0b4061","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.953994Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:ba3b24c875ef625c4ca02e1deb20b8015db22cbad21020c37b90adc9e9e65146","observation_id":"017ef2a9-5520-4eb9-bdc4-f92641554c31","resolution":{"observed_at":"2026-08-15T21:11:03.367736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.345434Z","title":null,"venue":null,"work_id":"1cd1b422-42e2-4551-95eb-1494d96c94b5","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.958191Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:92457c9f31cd6b3c7341badaf9168ec4d3104e9f5b4ff028e874c85b60053153","observation_id":"91339d58-f620-44d4-8eb7-b59703b0d07a","resolution":{"observed_at":"2026-08-15T21:11:03.350181Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.328985Z","title":"Martin, Hans-Theo Normann, and T","venue":null,"work_id":"880a825f-4dc6-4839-8693-140a2986625c","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.962360Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e2734550fbb0b70f81c856c98f0a16a0b962f766a96875573f132bceae4a45f6","observation_id":"1a302ee4-fd9f-46ae-8fab-c55bc4736872","resolution":{"observed_at":"2026-08-15T21:11:03.333745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.313485Z","title":"Rusu, Kieran Milan, John Quan, Tiago Ramalho, Agnieszka Grabska-Barwinska, et al","venue":null,"work_id":"d514e000-10d5-4f79-b13b-835ac92a4b4b","year":2017},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.966629Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:64ef772e3ca30ad5cd773d3ae6ba5bd1134c4074f730720604740fc5f42c8e1c","observation_id":"71a8d99b-1365-4a5b-ac7b-cce9c7fceaf8","resolution":{"observed_at":"2026-08-15T21:11:03.318008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03341","last_updated":"2024-06-26T14:11:53Z","snapshot_observed_at":"2026-08-13T19:18:46.730073Z","submitted_at":"2023-06-06T01:26:53Z","title":"Inference-Time Intervention: Eliciting Truthful Answers from a Language Model","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03341","snapshot_observed_at":"2026-08-15T21:11:01.971070Z","title":"Inference-time intervention: Eliciting truthful answers from a language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.971070Z"},"links":{"cited_paper":"/paper/2306.03341","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:8ec3b580dee0f7e52f6b4882e830bd197fc5815f4337a6a825e85f2a08204172","observation_id":"53ae654e-f604-43fa-bec1-7e4bbe48d9bf","resolution":{"observed_at":"2026-08-15T21:11:01.971070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05147","last_updated":"2024-08-19T07:51:05Z","snapshot_observed_at":"2026-08-15T15:50:24.074149Z","submitted_at":"2024-08-09T16:06:42Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05147","snapshot_observed_at":"2026-08-15T21:11:01.975880Z","title":"Gemma scope: Open sparse autoencoders everywhere all at once on gemma 2, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.975880Z"},"links":{"cited_paper":"/paper/2408.05147","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:afd87288f89c53bded5a31bdcffd12b196537682c17d17a00b1cf71d9a2e6ff8","observation_id":"accf6756-7ccc-46fa-8ac0-5a95cac9c7e1","resolution":{"observed_at":"2026-08-15T21:11:01.975880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.297754Z","title":"The mythos of model interpretability","venue":null,"work_id":"1da3f0f0-5edb-4d67-8d1f-43d5f4c3e2b8","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.980577Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:d44a11abead2638d598781ca122c021fff0e2af5cd07e9473b88d25665e5d5c6","observation_id":"e1e40780-c265-4c8d-b588-0f4db9b277d5","resolution":{"observed_at":"2026-08-15T21:11:03.302405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.989753Z","title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.989753Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:985f12b7f804488919109978059e0ace5334974e155652474eff4d728b5c8fc5","observation_id":"4fc9c058-9b7f-4030-b591-6b05a8f26e5f","resolution":{"observed_at":"2026-08-15T21:11:01.989753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05241","last_updated":"2024-10-30T18:37:57Z","snapshot_observed_at":"2026-08-16T13:28:20.917188Z","submitted_at":"2024-08-05T20:49:48Z","title":"Large Model Strategic Thinking, Small Model Efficiency: Transferring Theory of Mind in Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05241","snapshot_observed_at":"2026-08-15T21:11:01.994175Z","title":"Large model strategic thinking, small model efficiency: Transferring theory of mind in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.994175Z"},"links":{"cited_paper":"/paper/2408.05241","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:030cdc4113011a8ce88fafe08fb9218908fcd6c5b427a8aaf7dc5b402dc4c134","observation_id":"3cdffd23-9790-4415-81f6-49a0963613c8","resolution":{"observed_at":"2026-08-15T21:11:01.994175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.254557Z","title":"Linguistic regularities in continuous space word representations","venue":null,"work_id":"b53c9197-1afd-4023-96d6-b5b91c54f5d5","year":2013},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.003164Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:14c80c7581d4a59132af3b2eebacb274012a55c1410a42af5bcfba56cff8e605","observation_id":"c86296d2-ed87-49d3-b9f2-78d6f82f7afa","resolution":{"observed_at":"2026-08-15T21:11:03.259770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06196","last_updated":"2025-03-23T14:51:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-09T05:37:09Z","title":"Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06196","snapshot_observed_at":"2026-08-15T21:11:02.008126Z","title":"Large language models: A survey, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.008126Z"},"links":{"cited_paper":"/paper/2402.06196","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:55c09564f65472efadb3616a54763b1b3db9c37a2f326b50084d7175c664a6d4","observation_id":"a59a7951-c2e9-43c9-a557-199380bc4da6","resolution":{"observed_at":"2026-08-15T21:11:02.008126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.013053Z","title":"A strategy of win-stay, lose-shift that outperforms tit-for-tat in the prisoner’s dilemma game","venue":null,"work_id":null,"year":1993},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.013053Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:297bcf3d8af88404e6253ccda686701fe1cc383cab173314d7ae5f2310a35c5f","observation_id":"fae7cda6-218c-4394-8746-aa9f8ca079ab","resolution":{"observed_at":"2026-08-15T21:11:02.013053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-15T21:11:02.018166Z","title":"Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.018166Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:29f0349e400225acc354b33d713e013bbc145fcfe5d4df58884d2286ca59639b","observation_id":"4826b429-d15b-417a-b168-ff52c95b5cbc","resolution":{"observed_at":"2026-08-15T21:11:02.018166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"shsconf/2023178","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.602490Z","title":"Cooperation: A systematic review of how to enable agent to circumvent the prisoner’s dilemma","venue":null,"work_id":"4c42dc09-38e2-43a4-8bfd-3dd064e67328","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.023478Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:d385d9ad1c6e54fb67b436b766c33b215bb54555206632484457ba10c5292d27","observation_id":"014ab367-984a-479d-83c0-103ef33837a1","resolution":{"observed_at":"2026-08-15T21:11:02.613607Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03442","last_updated":"2023-08-06T00:21:19Z","snapshot_observed_at":"2026-08-16T03:49:04.888755Z","submitted_at":"2023-04-07T01:55:19Z","title":"Generative Agents: Interactive Simulacra of Human Behavior","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03442","snapshot_observed_at":"2026-08-15T21:11:02.028670Z","title":"Wang, Linxi Wang, Alex Wang, Allie He, Qian Liao, David Kempe, et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.028670Z"},"links":{"cited_paper":"/paper/2304.03442","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:764da072f0c834989483916a21b759a5cbcbee718e1e8544ee3d02dc32cd55d5","observation_id":"7a05f4db-7afc-4148-a98b-9ef26cb1000d","resolution":{"observed_at":"2026-08-15T21:11:02.028670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11871","last_updated":"2025-05-21T06:52:31Z","snapshot_observed_at":"2026-08-18T09:21:42.244522Z","submitted_at":"2024-10-09T12:06:43Z","title":"TinyClick: Single-Turn Agent for Empowering GUI Automation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11871","snapshot_observed_at":"2026-08-15T21:11:02.033414Z","title":"Tinyclick: Single-turn agent for empowering gui automation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.033414Z"},"links":{"cited_paper":"/paper/2410.11871","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:962e30645ac09c1a5f31691e29fb4c8b0fea0e76a9ab99e6a671c7a77f0669f4","observation_id":"de77e093-1471-462a-b0a6-a927f6729dc5","resolution":{"observed_at":"2026-08-15T21:11:02.033414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.226853Z","title":null,"venue":null,"work_id":"4b071fc5-ec94-4935-ba55-1533214252dd","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.038063Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:c97d6f66f4d5f927feb49290a840b83e350f38be2872f97ae5f41b59668728d5","observation_id":"73a4ef3f-38c0-44df-85f6-445319704ac3","resolution":{"observed_at":"2026-08-15T21:11:03.231251Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.211816Z","title":"Effect of private deliberation: Deception of large language models in game play","venue":null,"work_id":"b2f8b616-0fa5-494f-b414-6d88986d72fd","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.042680Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:fdc535878cba31685465388b0ccf7b803192e2dcf593801266a8bdd209cfed83","observation_id":"35196dd2-37eb-4576-a6c2-b40e2da03ea4","resolution":{"observed_at":"2026-08-15T21:11:03.216654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-16T07:06:07.822225Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-15T21:11:02.047108Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.047108Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:13af76d8492b66712c2129207adc1a23c693354c2dde509c3a572ee6a377edce","observation_id":"a70cb5aa-ff92-409f-b95a-3405b51bbf27","resolution":{"observed_at":"2026-08-15T21:11:02.047108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.051703Z","title":"A primer in BERTology: What we know about how BERT works","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.051703Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:5ac12d1a477e3668787213a4bc49aad5bd0d9267c7da646c01809819225c8efa","observation_id":"d5a5b1b9-7b0f-427c-973e-a741a2cd28cd","resolution":{"observed_at":"2026-08-15T21:11:02.051703Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.196936Z","title":"Research priorities for robust and beneficial artificial intelligence","venue":null,"work_id":"ca372a25-e789-4ef9-bfa3-87cbc0fdd2b1","year":2015},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.055985Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:bc0950e2b22be52f339bce528904fe053144ca4d6b4b42198a945f05dd705b61","observation_id":"cb2fe07f-6a82-4fb7-bc5a-7b4cefa20efd","resolution":{"observed_at":"2026-08-15T21:11:03.201675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.180346Z","title":null,"venue":null,"work_id":"7a0f9880-038a-435d-9bc7-149393ee1d1e","year":2019},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.060734Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:06ae9f13f86fe9b3b025d30a012e52163d5d45bb053ccfe19db259438902ef0a","observation_id":"7a2cf6c4-3cb3-42c1-a6c5-e33c3fd83511","resolution":{"observed_at":"2026-08-15T21:11:03.185336Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.04761","last_updated":"2023-02-09T16:49:57Z","snapshot_observed_at":"2026-07-06T14:50:07.491434Z","submitted_at":"2023-02-09T16:49:57Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.04761","snapshot_observed_at":"2026-08-15T21:11:02.064856Z","title":"Toolformer: Language models can teach themselves to use tools","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.064856Z"},"links":{"cited_paper":"/paper/2302.04761","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:43035a3f85fa6fd8654e66fc384379840dafb90e5693cfb3b0269ac1d1d87edd","observation_id":"95101970-bd83-42b3-bfc5-703b311591ce","resolution":{"observed_at":"2026-08-15T21:11:02.064856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.165755Z","title":"An evolutionary model of personality traits related to cooperative behavior using a large language model","venue":null,"work_id":"eeabdb4c-a419-4a55-ac68-df47d18dd88e","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.069111Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:98411265d2b0bd574b0f2e186264c2fe45f62821d4f1dd2649d16fddac1baa63","observation_id":"8f6167c9-4a5f-46af-b08b-065532a5c700","resolution":{"observed_at":"2026-08-15T21:11:03.170463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s11948-022-00392-3","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.183843Z","title":"A comparative analysis of the definitions of autonomous weapons systems","venue":null,"work_id":"65909bda-98a2-459d-9909-612e9bd6f539","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.073714Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e5412951b20370d0463f89fbd49778862e15752cbb9f85660d49da1fa07cea64","observation_id":"0b622b1d-44e4-43db-bc95-870dd9ca5525","resolution":{"observed_at":"2026-08-15T21:11:02.190554Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-15T21:11:02.078502Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.078502Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:ef023ec26272247cba1e30888ad222f5263ac07bbe407057a7f92bd3651a66ca","observation_id":"54cdc25e-dcbe-4a06-a3ca-a0b083ec66e7","resolution":{"observed_at":"2026-08-15T21:11:02.078502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-15T21:11:02.082737Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.082737Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:bccf4597c16c569626c3d43a0564d5d741769cf5f7484404b07a6292e59b3add","observation_id":"e70192c0-2fa3-4d26-9e99-25bce4521ab8","resolution":{"observed_at":"2026-08-15T21:11:02.082737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.087316Z","title":"Scaling monosemanticity: Extracting interpretable features from claude 3 sonnet","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.087316Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:81b313134bdab422d32cfc76fc0bacdb0c5faa703cf65bd43b18b8b4bd38a34b","observation_id":"625f59c6-0eef-406d-9cf0-d8c92c8ef449","resolution":{"observed_at":"2026-08-15T21:11:02.087316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.138657Z","title":"Moral alignment for llm agents","venue":null,"work_id":"6eaaf476-88be-45c4-b4a1-e6404ca5891b","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.092060Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:906818dbbc70e3879e2a181548bfe8f74882d3938cb73541599c2cbf3a054fad","observation_id":"7b2ebf77-d4d1-47eb-846b-c88fd4d56a1d","resolution":{"observed_at":"2026-08-15T21:11:03.143335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-15T21:11:02.098429Z","title":"Llama: Open and efficient foundation language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.098429Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:9f623e8e9a072adf447a5a14f7eca499dd093dccf882bcaeb1727a7d52620784","observation_id":"f523afad-184f-41ef-9a0e-57e19a11c506","resolution":{"observed_at":"2026-08-15T21:11:02.098429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-15T21:11:02.104902Z","title":"Llama 2: Open foundation and fine-tuned chat models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.104902Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:4c33d00608e1ce971047d87195ee193cf4016570ffae541f6b20d15bf4624a9c","observation_id":"82ac90b8-2d66-452e-9693-66c10e7d314b","resolution":{"observed_at":"2026-08-15T21:11:02.104902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1804.07461","last_updated":"2019-02-22T23:53:34Z","snapshot_observed_at":"2026-08-16T09:50:11.319379Z","submitted_at":"2018-04-20T06:35:04Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.07461","snapshot_observed_at":"2026-08-15T21:11:02.109561Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.109561Z"},"links":{"cited_paper":"/paper/1804.07461","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e029072b6d52231a090526021bc31127e9118ddd6905f84356cab8974aa8e6d2","observation_id":"20ce1bc9-e4d8-4904-9888-9de2d34c5479","resolution":{"observed_at":"2026-08-15T21:11:02.109561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.122837Z","title":"Mobile-agent: Autonomous multi-modal mobile device agent with visual perception,","venue":null,"work_id":"4964065f-8338-44e6-a00d-d8bbd9ae04dc","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.114293Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:06ded670fc70774e03b71908a19465a65a730496b4487509c9cf58a4dc9da1ca","observation_id":"f2942ef8-188f-4e1f-a44b-0747ba8e3471","resolution":{"observed_at":"2026-08-15T21:11:03.127797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.107500Z","title":"Dai, and Quoc V Le","venue":null,"work_id":"2f54aaa0-ecf9-4fa9-a849-fb3d0f2f63ab","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.123585Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:98b65b6755c1cd5e99a196894b4c7fb0be02f7390972e270c05072639c72cf15","observation_id":"e8c5f846-e91b-401e-80a4-0fd5d73e1194","resolution":{"observed_at":"2026-08-15T21:11:03.112312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-08-13T07:04:41.220509Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-15T21:11:02.128289Z","title":"Chain-of-thought prompting elicits reasoning in large language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.128289Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:7490b5daedf04105ae22a7ff91c365e9e41771b371ce4fa4c6f8f95f34c55d7f","observation_id":"9aac0e58-4508-4dce-b103-c860ac6c28f4","resolution":{"observed_at":"2026-08-15T21:11:02.128289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-15T21:11:02.133411Z","title":"ReAct: Synergizing reasoning and acting in language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.133411Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:8dbea1b06f143813b8530b029a8cb487b07d652c28985c523ae5820d38529b35","observation_id":"1cea8ac7-27e1-49cf-b37d-d6744ff11b68","resolution":{"observed_at":"2026-08-15T21:11:02.133411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13771","last_updated":"2026-07-05T07:51:04Z","snapshot_observed_at":"2026-08-16T14:32:38.574557Z","submitted_at":"2023-12-21T11:52:45Z","title":"AppAgent: Multimodal Agents as Smartphone Users","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13771","snapshot_observed_at":"2026-08-15T21:11:02.140648Z","title":"Appagent: Multimodal agents as smartphone users, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.140648Z"},"links":{"cited_paper":"/paper/2312.13771","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:66741ae45a9a65e21aa48371f7f22f9abb8c9a6788273da5048a51030e90c16d","observation_id":"647ac5ed-565f-4692-8bee-b21bc1834a89","resolution":{"observed_at":"2026-08-15T21:11:02.140648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16158","last_updated":"2024-04-18T06:53:38Z","snapshot_observed_at":"2026-08-16T13:42:20.592010Z","submitted_at":"2024-01-29T13:46:37Z","title":"Mobile-Agent: Autonomous Multi-Modal Mobile Device Agent with Visual Perception","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16158","snapshot_observed_at":"2026-08-15T21:11:02.118923Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.118923Z"},"links":{"cited_paper":"/paper/2401.16158","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:307dcd5fdf32fe62d9753454a2af0106f64a8cd7337be5dc3aef7d8b40a22429","observation_id":"33d30244-b423-4420-bcc9-fbd665b74b6d","resolution":{"observed_at":"2026-08-15T21:11:02.118923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01405","last_updated":"2025-03-03T06:14:14Z","snapshot_observed_at":"2026-07-06T16:26:38.284922Z","submitted_at":"2023-10-02T17:59:07Z","title":"Representation Engineering: A Top-Down Approach to AI Transparency","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01405","snapshot_observed_at":"2026-08-15T21:11:02.149478Z","title":"Byun, Zifan Wang, Alex Mallen, Steven Basart, Sanmi Koyejo, Dawn Song, Matt Fredrikson, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.149478Z"},"links":{"cited_paper":"/paper/2310.01405","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:b8bcc30508eac4338d3e2c2262d1a9c3ecdb760930a9ffd0df35be9ed84fc175","observation_id":"10bb76ee-bc2d-40f4-babf-40043502314b","resolution":{"observed_at":"2026-08-15T21:11:02.149478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11436","last_updated":"2024-06-07T04:52:29Z","snapshot_observed_at":"2026-08-16T14:59:03.737515Z","submitted_at":"2023-09-20T16:12:32Z","title":"You Only Look at Screens: Multimodal Chain-of-Action Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.11436","snapshot_observed_at":"2026-08-15T21:11:02.145319Z","title":"You only look at screens: Multimodal chain-of-action agents, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.145319Z"},"links":{"cited_paper":"/paper/2309.11436","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:729fe8725ee380b78bd5372858cdb5fd008b8ca2b39ae0883b93f3b629fdf8fa","observation_id":"e1633c79-0375-4295-bfa3-8325c644a49a","resolution":{"observed_at":"2026-08-15T21:11:02.145319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.984893Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.984893Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:977eba11d180dca9efb9b3b0e9c89a04c2ec480ef3caff4d518eb103eea92809","observation_id":"d843c580-4512-4d8f-a3a6-6b41724a5ec6","resolution":{"observed_at":"2026-08-15T21:11:01.984893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-15T21:11:01.925637Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.925637Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:c555218d0202e2b5159da1a3bd738146b202bdbcfbd361368474ba9320e7434f","observation_id":"671ed4ec-d1ad-4d99-97f8-e0d15322b52b","resolution":{"observed_at":"2026-08-15T21:11:01.925637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.469124Z","title":null,"venue":null,"work_id":"9cbe17c8-0558-45a5-81b9-224ec7f22f69","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.866455Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:3e70126129b6dcab211899849f5d49426ce28d479d894d4b9db18496c1c13dad","observation_id":"6b1e14ac-f124-489b-84f4-00a88d7ea70d","resolution":{"observed_at":"2026-08-15T21:11:03.474249Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.271093Z","title":null,"venue":null,"work_id":"9b87a092-16aa-452f-b667-1928c6c93cf8","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.998638Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:ca80af219653425a6bab904a173c93769f0c79c273b411f490a2f3f17fd8967e","observation_id":"98fbd528-7341-4dfe-92d5-440307b35277","resolution":{"observed_at":"2026-08-15T21:11:03.276283Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-18T11:09:54.533839Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems"},"reference_resolution":{"displayed":77,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":49,"verified_exact":2,"verified_fuzzy":24},"total_outbound_references":77},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 77 of 77 outbound references and 3 inbound Pith citation observations for arXiv:2505.10670."}