{"as_of":"2026-08-11T13:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:145e6eaefad8fea39def7abeec2784e665de5b40e9e1fe5ea3596e351ce7a23e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":57,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:10:19.011861Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":14,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2405.00592","last_updated":"2025-06-30T15:11:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-01T15:59:00Z","title":"Scaling and renormalization in high-dimensional regression","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-24T01:54:48.781227Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2405.00592"},"observation_digest":"sha256:f30d395f489bc993f62af907787cd35d24b6643570ea1869ce1b576e8c1881d3","observation_id":"e47d2fa6-25b8-4557-8bcc-89e4f3513b11","resolution":{"observed_at":"2026-05-24T01:55:55.089386Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2406.18495","last_updated":"2024-12-09T20:21:56Z","snapshot_observed_at":"2026-08-06T04:32:59.039501Z","submitted_at":"2024-06-26T16:58:20Z","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T16:25:14.744887Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2406.18495"},"observation_digest":"sha256:d61330daedc7c39e23de87f995e221065c7ff419fb5cfe4dd4ace5a2f5b4c73f","observation_id":"07a1da5b-b4f0-44e8-b987-1a99bf9035de","resolution":{"observed_at":"2026-05-17T16:25:14.801617Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-11T13:10:19.011861Z","title":"S.; Jenner, E.; Casper, S.; Sourbut, O.; et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13471","last_updated":"2024-12-18T03:36:08Z","snapshot_observed_at":"2026-08-11T13:03:53.336220Z","submitted_at":"2024-12-18T03:36:08Z","title":"Gradual Vigilance and Interval Communication: Enhancing Value Alignment in Multi-Agent Debates","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T13:10:19.011861Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2412.13471"},"observation_digest":"sha256:e8da4ae307c25e8e71fb24c963ce9ab38700f040d7dbe077c0741da69a8c4836","observation_id":"9390ee86-a5d9-4451-bedf-148ca9ffb44d","resolution":{"observed_at":"2026-08-11T13:10:19.011861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-11T12:47:48.277131Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13821","last_updated":"2024-12-18T13:10:35Z","snapshot_observed_at":"2026-08-11T12:43:13.784843Z","submitted_at":"2024-12-18T13:10:35Z","title":"Towards Responsible Governing AI Proliferation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T12:47:48.277131Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2412.13821"},"observation_digest":"sha256:845448562430c0b3d2973684b6df6909aa748159977ee61695082ac08f9cf2c7","observation_id":"94a84c6a-a27e-4769-a43b-42f026ef23e9","resolution":{"observed_at":"2026-08-11T12:47:48.277131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-11T10:42:47.257659Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16355","last_updated":"2025-04-02T19:56:19Z","snapshot_observed_at":"2026-08-11T10:37:35.767116Z","submitted_at":"2024-12-20T21:34:43Z","title":"Social Science Is Necessary for Operationalizing Socially Responsible Foundation Models","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T10:42:47.257659Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2412.16355"},"observation_digest":"sha256:1b39a25f23cfd35caa350b21157e99beab99b9111e7b5cd676d91f8aba7abad1","observation_id":"e5d1ffb1-306f-4bb6-a7dd-2e27061568e2","resolution":{"observed_at":"2026-08-11T10:42:47.257659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T22:53:20.428520Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.00517","last_updated":"2024-12-31T16:01:25Z","snapshot_observed_at":"2026-08-10T22:46:48.760365Z","submitted_at":"2024-12-31T16:01:25Z","title":"A Method for Enhancing the Safety of Large Model Generation Based on Multi-dimensional Attack and Defense","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T22:53:20.428520Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.00517"},"observation_digest":"sha256:14830ca7475bebc4bfd948e4bfa299e873f196c32881293cc97ce50b364da091","observation_id":"9c9485fd-3d9b-48a2-a535-f7d52dd85d64","resolution":{"observed_at":"2026-08-10T22:53:20.428520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T22:35:26.875258Z","title":"Valérie JV Broers, Céline De Breucker, Stephan Van den Broucke, and Olivier Luminet","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.02018","last_updated":"2025-01-02T15:15:38Z","snapshot_observed_at":"2026-08-10T22:28:18.009629Z","submitted_at":"2025-01-02T15:15:38Z","title":"Safeguarding Large Language Models in Real-time with Tunable Safety-Performance Trade-offs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T22:35:26.875258Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.02018"},"observation_digest":"sha256:3e1246841a09bef1a338e896b8e1d57b95e1c80d67d7756fcb84ab24e6fabbeb","observation_id":"b11812df-a75e-499f-a9e1-35055655b0f2","resolution":{"observed_at":"2026-08-10T22:35:26.875258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T21:18:11.730920Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05398","last_updated":"2025-01-09T17:47:34Z","snapshot_observed_at":"2026-08-10T23:14:25.374382Z","submitted_at":"2025-01-09T17:47:34Z","title":"Mechanistic understanding and validation of large AI models with SemanticLens","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T21:18:11.730920Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.05398"},"observation_digest":"sha256:f516f37b5bcd18a601d0bd1837458e889e11bb23b4a9c54341572774ca4f8c6f","observation_id":"edaf3004-9b57-4970-a514-98992a9076e3","resolution":{"observed_at":"2026-08-10T21:18:11.730920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T20:35:52.714194Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.07959","last_updated":"2025-02-01T09:30:34Z","snapshot_observed_at":"2026-08-11T11:45:00.467395Z","submitted_at":"2025-01-14T09:23:30Z","title":"Self-Instruct Few-Shot Jailbreaking: Decompose the Attack into Pattern and Behavior Learning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:35:52.714194Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.07959"},"observation_digest":"sha256:8bc2317514fd55e72888de961338a54fbed17410aaa0c2d5ed8e08280b7198bd","observation_id":"6bb033e1-ce8e-48c9-a83d-809d6234bd9a","resolution":{"observed_at":"2026-08-10T20:35:52.714194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T20:15:11.764071Z","title":"Anwar, A","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.09254","last_updated":"2025-01-16T02:43:44Z","snapshot_observed_at":"2026-08-11T01:59:55.940834Z","submitted_at":"2025-01-16T02:43:44Z","title":"Clone-Robust AI Alignment","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:15:11.764071Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.09254"},"observation_digest":"sha256:ca81c0d38b9301ce6bf01e972346be7a047ab2dc53569275178bdb1b8015e76d","observation_id":"4a0ec520-0d0d-45c1-97ad-35ebf5e33ec5","resolution":{"observed_at":"2026-08-10T20:15:11.764071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T19:55:50.488108Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09620","last_updated":"2025-05-29T02:21:03Z","snapshot_observed_at":"2026-08-10T19:47:49.643885Z","submitted_at":"2025-01-16T16:00:37Z","title":"Beyond Reward Hacking: Causal Rewards for Large Language Model Alignment","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T19:55:50.488108Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.09620"},"observation_digest":"sha256:63bbfd91a5a196412100447c452656203dddffbffbd89bd82e0d5909b284fa20","observation_id":"e88b23e9-baaa-4345-a6a3-3101215e6698","resolution":{"observed_at":"2026-08-10T19:55:50.488108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T17:58:17.922278Z","title":"Foundational challenges in assuring alignment and safety of large langua ge models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.11739","last_updated":"2025-01-22T15:09:02Z","snapshot_observed_at":"2026-08-10T17:52:46.139618Z","submitted_at":"2025-01-20T20:54:06Z","title":"Episodic memory in AI agents poses risks that should be studied and mitigated","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T17:58:17.922278Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.11739"},"observation_digest":"sha256:055294e322a0eed6a59817975ff0c75c7c37866766de20d09bb7450495303887","observation_id":"9a148bf4-fa45-45ed-9888-668ed4c51d37","resolution":{"observed_at":"2026-08-10T17:58:17.922278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T14:51:58.699740Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.14940","last_updated":"2026-06-28T22:17:10Z","snapshot_observed_at":"2026-08-10T14:57:28.038497Z","submitted_at":"2025-01-24T21:55:14Z","title":"CASE-Bench: Context-Aware SafEty Benchmark for Large Language Models","version":4},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-10T14:51:58.699740Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2501.14940"},"observation_digest":"sha256:4d3e6ab6e1b8aa87070742c70b80520fe0e5e44d5e43cee4aa2c7fdc26d48845","observation_id":"926cd96f-cc6f-4f4b-ae9a-489852782b6f","resolution":{"observed_at":"2026-08-10T14:51:58.699740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-09T14:48:34.657585Z","title":"S., Jenner, E., Casper, S., Sourbut, O., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01635","last_updated":"2025-02-03T18:59:13Z","snapshot_observed_at":"2026-08-10T10:52:49.507908Z","submitted_at":"2025-02-03T18:59:13Z","title":"The AI Agent Index","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-09T14:48:34.657585Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2502.01635"},"observation_digest":"sha256:efabcbe94fe3c85e1d3cfbc4aa997bb63dbc40855f42eb53565238e0ea6f96d0","observation_id":"2e708e35-c5da-4e6d-831e-ba710e2645fa","resolution":{"observed_at":"2026-08-09T14:48:34.657585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-09T05:09:16.102465Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04376","last_updated":"2025-02-05T16:25:43Z","snapshot_observed_at":"2026-08-10T00:16:25.433510Z","submitted_at":"2025-02-05T16:25:43Z","title":"MEETING DELEGATE: Benchmarking LLMs on Attending Meetings on Our Behalf","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-09T05:09:16.102465Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2502.04376"},"observation_digest":"sha256:50dee1d6d9b17b59711e807472f82d7eda3caf037bc78f01e68280058b8d6020","observation_id":"0d821892-22fe-43fa-9cf9-a296911e69e9","resolution":{"observed_at":"2026-08-09T05:09:16.102465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-08T12:25:30.511215Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07557","last_updated":"2025-02-11T13:50:50Z","snapshot_observed_at":"2026-08-09T03:24:13.351555Z","submitted_at":"2025-02-11T13:50:50Z","title":"JBShield: Defending Large Language Models from Jailbreak Attacks through Activated Concept Analysis and Manipulation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T12:25:30.511215Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2502.07557"},"observation_digest":"sha256:b97c4ecba971dcd3f3fe332b12288c322f62092f80329b934ec8e5ea19cb211a","observation_id":"67090237-d39b-4ec6-85bb-953388fac2cc","resolution":{"observed_at":"2026-08-08T12:25:30.511215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T19:45:20.088578Z","title":"Foundational challenges in assuring alignment and safety of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.14881","last_updated":"2025-02-14T08:42:43Z","snapshot_observed_at":"2026-08-07T22:02:21.919237Z","submitted_at":"2025-02-14T08:42:43Z","title":"A Survey of Safety on Large Vision-Language Models: Attacks, Defenses and Evaluations","version":1},"reference_index":167,"source":"pdf_text","source_observed_at":"2026-08-07T19:45:20.088578Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2502.14881"},"observation_digest":"sha256:99275a2e731aa2f03e34070077cd3140976ba8f022c920f9c15a7bdf92886e6d","observation_id":"276b0fa3-77a0-4277-a525-fee2aa77b9ef","resolution":{"observed_at":"2026-08-07T19:45:20.088578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T14:30:04.428012Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18807","last_updated":"2025-05-24T17:41:47Z","snapshot_observed_at":"2026-08-11T07:37:54.093177Z","submitted_at":"2025-05-24T17:41:47Z","title":"Mitigating Deceptive Alignment via Self-Monitoring","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:04.428012Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2505.18807"},"observation_digest":"sha256:b6a8e1278710d10dd84e8943a5b9f16dd3dfdec281c56cae1da8d2d3c098d8e0","observation_id":"0c2bc9d3-9b3a-4bd8-a2df-7f37c0a94918","resolution":{"observed_at":"2026-08-07T14:30:04.428012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2505.19241","last_updated":"2026-05-14T18:01:35Z","snapshot_observed_at":"2026-08-02T15:42:14.270446Z","submitted_at":"2025-05-25T17:42:52Z","title":"ActiveDPO: Active Direct Preference Optimization for Sample-Efficient Alignment","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T01:06:19.756032Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2505.19241"},"observation_digest":"sha256:01c4a21528f68083a4b39b0da96f928809876dd3314a95bb34cf76ca2a719716","observation_id":"eaaae2ea-aa11-4887-9fc5-b7df25e0b9b0","resolution":{"observed_at":"2026-05-22T01:10:51.554189Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T14:11:30.354244Z","title":"[Anwar et al., 2024] Usman Anwar, Abulhair Saparov, Javier Rando, Daniel Paleka, Miles Turpin, Peter Hase, Ekdeep Singh Lubana, Erik Jenner, Stephen Casper, Oliver Sourbut, et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19743","last_updated":"2025-08-16T11:40:47Z","snapshot_observed_at":"2026-08-09T04:25:32.618058Z","submitted_at":"2025-05-26T09:24:36Z","title":"Token-level Accept or Reject: A Micro Alignment Approach for Large Language Models","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:30.354244Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2505.19743"},"observation_digest":"sha256:8e8ebe76841b713dbd4b29395b788d3dbed751fc1a2c484cdd3ea8c318b9aff0","observation_id":"6604e4c7-4876-4fbf-a6c5-4753b52ba655","resolution":{"observed_at":"2026-08-07T14:11:30.354244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T12:40:13.874731Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24119","last_updated":"2025-05-30T01:32:44Z","snapshot_observed_at":"2026-08-07T23:14:32.953063Z","submitted_at":"2025-05-30T01:32:44Z","title":"The State of Multilingual LLM Safety Research: From Measuring the Language Gap to Mitigating It","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T12:40:13.874731Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2505.24119"},"observation_digest":"sha256:9901a98b403ac44fd3ed5b21a0d29f2e48a47b066de4eeb79127e7813edc3467","observation_id":"0d1cbc67-e62a-4ab9-8b73-d7ef4d75dfb6","resolution":{"observed_at":"2026-08-07T12:40:13.874731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T13:09:58.104897Z","title":"S., Jenner, E., Casper, S., Sourbut, O., Edelman, B","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00047","last_updated":"2025-05-28T16:52:44Z","snapshot_observed_at":"2026-08-07T13:01:23.937258Z","submitted_at":"2025-05-28T16:52:44Z","title":"Risks of AI-driven product development and strategies for their mitigation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T13:09:58.104897Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2506.00047"},"observation_digest":"sha256:4566c5b6bebda372a15b7c70c718d622b61aec60e462882906e42b05e9cd7eac","observation_id":"75c56f71-475a-42c5-8fca-4ab6831a08d1","resolution":{"observed_at":"2026-08-07T13:09:58.104897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T11:17:28.039432Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02996","last_updated":"2025-06-03T15:31:00Z","snapshot_observed_at":"2026-08-08T03:16:36.105819Z","submitted_at":"2025-06-03T15:31:00Z","title":"Linear Spatial World Models Emerge in Large Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:17:28.039432Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2506.02996"},"observation_digest":"sha256:8f7da9baacea42319906bb347d72f489cb1c7af859d8c7cbc090da4def562138","observation_id":"6ffa341c-45b3-4e5d-9adc-c2fa387f1b01","resolution":{"observed_at":"2026-08-07T11:17:28.039432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T10:54:57.913128Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04018","last_updated":"2026-06-22T12:47:48Z","snapshot_observed_at":"2026-08-09T08:53:39.548570Z","submitted_at":"2025-06-04T14:46:47Z","title":"AgentMisalignment: Measuring the Propensity for Misaligned Behaviour in LLM-Based Agents","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T10:54:57.913128Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2506.04018"},"observation_digest":"sha256:3483d5b14f75ef14f80e8eeda0194bb93319c49fa90d58ce1405927ed6577a60","observation_id":"e14fbc27-e4c3-4582-9af6-2e467cb1714f","resolution":{"observed_at":"2026-08-07T10:54:57.913128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-07T00:32:36.451035Z","title":"Foundational challenges in assuring alignment and safety of large language models.arXiv preprint arXiv:2404.09932, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13666","last_updated":"2025-06-16T16:24:31Z","snapshot_observed_at":"2026-08-07T00:25:36.770856Z","submitted_at":"2025-06-16T16:24:31Z","title":"We Should Identify and Mitigate Third-Party Safety Risks in MCP-Powered Agent Systems","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:32:36.451035Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2506.13666"},"observation_digest":"sha256:86ffa872444dbf1d303ded7bef9eef6a4720ef946e84030cd405bfae48f2ebd3","observation_id":"e62633f9-e83d-4a72-b74f-41265a3c4fd3","resolution":{"observed_at":"2026-08-07T00:32:36.451035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-06T23:49:13.094114Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16078","last_updated":"2025-06-19T07:03:05Z","snapshot_observed_at":"2026-08-07T07:58:06.769500Z","submitted_at":"2025-06-19T07:03:05Z","title":"Probing the Robustness of Large Language Models Safety to Latent Perturbations","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T23:49:13.094114Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2506.16078"},"observation_digest":"sha256:5aa64f71a5dbfba3f9eb40883c9498211b69aa258ca6cc12cde19405fe5d4365","observation_id":"28c35c6b-b441-4b2e-a413-138b5172816b","resolution":{"observed_at":"2026-08-06T23:49:13.094114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-06T21:58:22.466382Z","title":"Edelman, Zhaowei Zhang, Mario Günther, Anton Korinek, José Hernández-Orallo, Lewis Ham- mond, Eric J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22957","last_updated":"2025-08-27T20:38:04Z","snapshot_observed_at":"2026-08-10T19:02:05.215851Z","submitted_at":"2025-06-28T17:22:59Z","title":"Agent-to-Agent Theory of Mind: Testing Interlocutor Awareness among Large Language Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:58:22.466382Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2506.22957"},"observation_digest":"sha256:50140bf364e54771ee34fa236eab2c62cf4057d8359f5dccda098e348e0233ce","observation_id":"555ad9b2-e220-459a-b447-9abe1b5f31e0","resolution":{"observed_at":"2026-08-06T21:58:22.466382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-06T19:07:40.377039Z","title":"S., Jenner, E., Casper, S., Sourbut, O., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06434","last_updated":"2025-07-08T22:29:06Z","snapshot_observed_at":"2026-08-08T23:28:38.307914Z","submitted_at":"2025-07-08T22:29:06Z","title":"Deprecating Benchmarks: Criteria and Framework","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T19:07:40.377039Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2507.06434"},"observation_digest":"sha256:d2ba11e6fabafda71c3d7753ab0113c00eb7b94a73f4530a5a9450057eb0da76","observation_id":"d718aa0a-df81-4905-89ce-89b731533a5d","resolution":{"observed_at":"2026-08-06T19:07:40.377039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":292,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:4b62a311aa48b387e0e4c49bcf8401f50a2ff31ba0b033d4ba528095cf364210","observation_id":"3e01fb38-ff2f-4f15-9450-db40c886c0bf","resolution":{"observed_at":"2026-05-14T22:23:15.423464Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-06T12:22:27.592346Z","title":"First, we have shown in section 3 that the downsides of racing to AGI are much higher than proponents of AGI Racing hold","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.21839","last_updated":"2025-07-29T14:17:08Z","snapshot_observed_at":"2026-08-08T03:45:41.895034Z","submitted_at":"2025-07-29T14:17:08Z","title":"Against racing to AGI: Cooperation, deterrence, and catastrophic risks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T12:22:27.592346Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2507.21839"},"observation_digest":"sha256:2a25f426249c5d53d5650f76fbbfc4cbb8aa155bb31b3063920f92da795aed82","observation_id":"2e208163-7343-40e9-b35a-595ddf22ab0f","resolution":{"observed_at":"2026-08-06T12:22:27.592346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-06T11:34:24.795223Z","title":"Edelman, Zhaowei Zhang, Mario Günther, Anton Korinek, José Hernández-Orallo, Lewis Hammond, Eric J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.22617","last_updated":"2025-07-30T12:37:29Z","snapshot_observed_at":"2026-08-06T22:42:04.029387Z","submitted_at":"2025-07-30T12:37:29Z","title":"Hate in Plain Sight: On the Risks of Moderating AI-Generated Hateful Illusions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T11:34:24.795223Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2507.22617"},"observation_digest":"sha256:2d10e572c57eec4248e309b9b2170f66209b871213983e67a97d7c5c71a61da2","observation_id":"372bda71-ed0d-4395-82e4-c3299076289b","resolution":{"observed_at":"2026-08-06T11:34:24.795223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T20:07:02.902944Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11252","last_updated":"2025-08-15T06:42:00Z","snapshot_observed_at":"2026-08-08T07:13:37.375064Z","submitted_at":"2025-08-15T06:42:00Z","title":"Beyond Solving Math Quiz: Evaluating the Ability of Large Reasoning Models to Ask for Information","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-05T20:07:02.902944Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2508.11252"},"observation_digest":"sha256:7c2ccf986deeabe689606fc0b263f4f76e79e74eb776baadea3532d732fa4805","observation_id":"e0f6b72d-cb2b-44a1-a86b-39d5e21a3c38","resolution":{"observed_at":"2026-08-05T20:07:02.902944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T04:50:31.598124Z","title":"Foundational challenges in assuring alignment and safety of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05946","last_updated":"2025-09-07T06:46:03Z","snapshot_observed_at":"2026-08-07T19:21:09.227690Z","submitted_at":"2025-09-07T06:46:03Z","title":"Large Language Models for Next-Generation Wireless Network Management: A Survey and Tutorial","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T04:50:31.598124Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2509.05946"},"observation_digest":"sha256:ba21d889c453d09a6560754209272cd739da610420d1d2a434e9eaa1ae05dc00","observation_id":"ecc3fec7-e818-46c6-b563-51e46ebf87d9","resolution":{"observed_at":"2026-08-05T04:50:31.598124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2510.12826","last_updated":"2026-04-25T22:54:37Z","snapshot_observed_at":"2026-08-09T21:13:46.129374Z","submitted_at":"2025-10-11T04:42:29Z","title":"Scheming Ability in LLM-to-LLM Strategic Interactions","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T07:50:30.597108Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2510.12826"},"observation_digest":"sha256:8a150fb339021495bf0c2c06a5aa3cb02f303b4ee8a18342100412e24cf0dbc8","observation_id":"e8c68cc1-77b3-4e49-af8b-3d246cf85831","resolution":{"observed_at":"2026-05-18T07:51:03.750005Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-04T07:21:32.304935Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.26707","last_updated":"2026-07-15T04:49:02Z","snapshot_observed_at":"2026-08-07T05:29:04.731147Z","submitted_at":"2025-10-30T17:09:09Z","title":"Value Drifts: Tracing Value Alignment During LLM Post-Training","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-04T07:21:32.304935Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2510.26707"},"observation_digest":"sha256:269be5686782adc394a0d696912a8ec42d674b03e17bffd84dbfdccd142f5302","observation_id":"c691439b-6c02-4607-97ce-bce0e5225742","resolution":{"observed_at":"2026-08-04T07:21:32.304935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-03T04:59:05.071079Z","title":"URL https: //doi.org/10.48550/arXiv.2404.09932","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.04899","last_updated":"2026-06-02T15:48:30Z","snapshot_observed_at":"2026-08-07T15:08:25.314163Z","submitted_at":"2026-02-03T14:38:07Z","title":"Phantom Transfer: Data Poisoning can Survive Data-Level Defences","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T04:59:05.071079Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2602.04899"},"observation_digest":"sha256:b2adc9e912ebd0eee086dc0c3b35841a5a044d5456872199412080628f247501","observation_id":"6777e281-4e8f-406e-9a6a-71c8f5b6d78c","resolution":{"observed_at":"2026-08-03T04:59:05.071079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2602.07340","last_updated":"2026-05-21T07:39:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-07T03:46:33Z","title":"Revisiting Robustness for LLM Safety Alignment via Selective Geometry Control","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T11:17:03.104902Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2602.07340"},"observation_digest":"sha256:dea65fd80bdb8b3a1ec604f2f45c10ceecdc32fa19b250c565013f88f5d767ca","observation_id":"41cc4a35-8e10-49c1-a665-dadfb8cdaed7","resolution":{"observed_at":"2026-05-22T11:21:28.998865Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-02T23:21:31.165458Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.14095","last_updated":"2026-07-06T07:40:26Z","snapshot_observed_at":"2026-08-09T00:10:29.296236Z","submitted_at":"2026-02-15T11:05:18Z","title":"NEST: Nascent Encoded Steganographic Thoughts","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T23:21:31.165458Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2602.14095"},"observation_digest":"sha256:e00b0cc0f01e51a40ce16402d56f3da376eeab70787a7a4afa437488f3aacdbf","observation_id":"4bc4ceda-8734-4625-addb-79961c6c573b","resolution":{"observed_at":"2026-08-02T23:21:31.165458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2602.20102","last_updated":"2026-05-21T18:01:19Z","snapshot_observed_at":"2026-08-02T14:02:01.747985Z","submitted_at":"2026-02-23T18:19:46Z","title":"BarrierSteer: LLM Safety via Learning Barrier Steering","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-25T07:02:03.058731Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2602.20102"},"observation_digest":"sha256:6b73e69984610fd2180a5e3e65f5f2c5c5e8fd86a1bce642f253155785d25488","observation_id":"21fcd595-f29d-4fd0-8931-13f04dc59457","resolution":{"observed_at":"2026-05-25T07:05:26.712893Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2605.05115","last_updated":"2026-05-06T16:46:03Z","snapshot_observed_at":"2026-08-07T10:12:24.411121Z","submitted_at":"2026-05-06T16:46:03Z","title":"Manifold Steering Reveals the Shared Geometry of Neural Network Representation and Behavior","version":1},"reference_index":228,"source":"arxiv_source","source_observed_at":"2026-05-08T17:47:09.591001Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2605.05115"},"observation_digest":"sha256:9bb94038c1f45a48db19a98a2414c3947d1d80fe1a3aac9035101b732626889f","observation_id":"f1f45249-5e71-4b83-be97-9d66264b82b8","resolution":{"observed_at":"2026-05-11T17:16:06.543864Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2605.08405","last_updated":"2026-05-08T19:11:19Z","snapshot_observed_at":"2026-08-11T07:23:59.123455Z","submitted_at":"2026-05-08T19:11:19Z","title":"Belief or Circuitry? Causal Evidence for In-Context Graph Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T00:52:07.392800Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2605.08405"},"observation_digest":"sha256:a4e306b160a65aceaf4895e035c8eafbe18ef5ea8fce10d978392dfebb9d8bca","observation_id":"ffd801da-ca43-4ed6-b6f6-99f976febded","resolution":{"observed_at":"2026-05-12T08:41:23.962333Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2605.11161","last_updated":"2026-05-11T19:08:21Z","snapshot_observed_at":"2026-08-03T18:26:43.880438Z","submitted_at":"2026-05-11T19:08:21Z","title":"Interpretability Can Be Actionable","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-05-13T06:12:51.656452Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2605.11161"},"observation_digest":"sha256:fb2b57c582edc64feb9e1a0c0b8b8d49ec117f06e1356a47633a3a8ad7a51da9","observation_id":"2bcdda77-c8c6-4b91-8ba4-782e7a398082","resolution":{"observed_at":"2026-05-13T06:17:22.798229Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2605.12412","last_updated":"2026-05-12T17:09:41Z","snapshot_observed_at":"2026-08-11T04:19:07.969715Z","submitted_at":"2026-05-12T17:09:41Z","title":"Stories in Space: In-Context Learning Trajectories in Conceptual Belief Space","version":1},"reference_index":144,"source":"arxiv_source","source_observed_at":"2026-05-13T05:17:34.283917Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2605.12412"},"observation_digest":"sha256:20bc873c8ddae3342e1cf4bac07db77969166c8969386561966bc07707331fd4","observation_id":"cf4c3b87-5ed5-45d5-b964-48f9624777db","resolution":{"observed_at":"2026-05-13T05:27:19.282765Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2605.29249","last_updated":"2026-05-28T02:09:38Z","snapshot_observed_at":"2026-08-06T07:11:23.483060Z","submitted_at":"2026-05-28T02:09:38Z","title":"Prediction-Powered Inference Across Many Tasks for AI Evaluation & Social Science Research","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T06:00:59.614364Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2605.29249"},"observation_digest":"sha256:807c44b0094a469a5dbc05e86c0818be64a47f267bc9c4c3f29f9e3003bf747c","observation_id":"e4284353-edd1-4007-8b40-4521a20c3b61","resolution":{"observed_at":"2026-06-29T06:03:08.437075Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2605.30169","last_updated":"2026-07-02T23:58:01Z","snapshot_observed_at":"2026-08-01T18:48:30.510137Z","submitted_at":"2026-05-28T16:20:19Z","title":"Dissociative Identity: Language Model Agents Lack Grounding for Reputation Mechanisms","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T00:26:54.019256Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2605.30169"},"observation_digest":"sha256:77810c88e7fa455ba1197276dacef636b91efb23058ad74d573990c309c0a89d","observation_id":"b7dfc7f5-81bb-491b-ae4d-bf9f74b8ffa8","resolution":{"observed_at":"2026-06-29T00:32:53.047592Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2605.30454","last_updated":"2026-05-28T18:26:40Z","snapshot_observed_at":"2026-07-31T16:31:43.334849Z","submitted_at":"2026-05-28T18:26:40Z","title":"The Surface You Test Is Not the Surface That Breaks","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-29T06:37:19.674012Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2605.30454"},"observation_digest":"sha256:eed872f93a16b478cf41489f7eba017ce280ff88bcbe38d3b8581c07d43a788f","observation_id":"ca044699-010f-4031-944a-97e2aeaea091","resolution":{"observed_at":"2026-06-29T14:33:31.190683Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2606.01929","last_updated":"2026-06-09T14:23:44Z","snapshot_observed_at":"2026-08-08T09:05:32.069117Z","submitted_at":"2026-06-01T08:59:03Z","title":"VET: A Framework for Analyzing AI Discourse","version":2},"reference_index":166,"source":"arxiv_source","source_observed_at":"2026-06-28T14:41:17.714421Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2606.01929"},"observation_digest":"sha256:f0161cb83b093c61c79119ce2ee29d59b3a7df31bee5b8087800f4451520303f","observation_id":"ee43106b-6bdd-435c-b545-49b57857f71f","resolution":{"observed_at":"2026-07-01T23:06:20.479301Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2606.02211","last_updated":"2026-06-01T13:10:49Z","snapshot_observed_at":"2026-08-02T18:41:16.377453Z","submitted_at":"2026-06-01T13:10:49Z","title":"Consistency Training while Mitigating Obfuscation via Rate Matching","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-06-28T14:25:43.147442Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2606.02211"},"observation_digest":"sha256:ed7d619cdc8745d30b7344c2f1b3390a5696f9684af39290ce4203defffcf1eb","observation_id":"fe12e385-1113-407c-b812-b8418590317b","resolution":{"observed_at":"2026-06-28T14:32:18.165223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2606.03647","last_updated":"2026-06-02T13:39:15Z","snapshot_observed_at":"2026-07-06T23:43:50.943530Z","submitted_at":"2026-06-02T13:39:15Z","title":"Black-box, Adaptive, Efficient, Transferable, Harmful, Applicable... Attacks Are All You Need to Break LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T09:21:57.373862Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2606.03647"},"observation_digest":"sha256:81b1b34bd8211cdac01d44ca9e5d9c71ab4ae9dee8c85ea7b23cae0a3f990847","observation_id":"2e78d5e7-5ce1-4849-b17c-6a011716ecbf","resolution":{"observed_at":"2026-07-02T04:16:34.761998Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2606.07007","last_updated":"2026-06-05T07:52:43Z","snapshot_observed_at":"2026-08-04T06:36:27.490603Z","submitted_at":"2026-06-05T07:52:43Z","title":"A Geometric View for Understanding Concept Learning and Neuron Interpretation in Sparse Autoencoders","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-06-27T22:22:50.474397Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2606.07007"},"observation_digest":"sha256:9d6e517a7a7349cc17da3c98e3628bfd0bb1d4249914612f9d91cc8cbdce434d","observation_id":"46a39709-f2b5-41aa-892b-53ffb6583773","resolution":{"observed_at":"2026-07-02T16:47:09.934660Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2606.08381","last_updated":"2026-06-07T00:20:55Z","snapshot_observed_at":"2026-08-06T21:12:14.765265Z","submitted_at":"2026-06-07T00:20:55Z","title":"Auditing Proprietary Alignment in Large Language Models: A Comparative Framework Without a Ground-Truth Standard","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T19:04:07.735560Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2606.08381"},"observation_digest":"sha256:d8a1394125920ed6a4c41adfe4bbe3f1710995cbd5880908e353eae034812a44","observation_id":"dcbdf9e3-9228-4c45-bb78-1a1947a62269","resolution":{"observed_at":"2026-07-02T22:17:26.017226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2606.09388","last_updated":"2026-06-08T12:03:51Z","snapshot_observed_at":"2026-08-07T01:03:05.186184Z","submitted_at":"2026-06-08T12:03:51Z","title":"Distilling Safe LLM Systems via Soft Prompts for On Device Settings","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-06-27T17:15:51.375580Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2606.09388"},"observation_digest":"sha256:5c1413ef769113e1257e52f14d9367a4c5a7ae58632ccab149eeaf615d4a217b","observation_id":"5eb8546a-7b7b-4288-9804-0fd91738197e","resolution":{"observed_at":"2026-07-03T00:27:29.222546Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":"2404.09932","doi":"10.48550/arxiv.2404.09932","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anwar, U., Saparov, A., Rando, J., Paleka, D., Turpin, M., Hase, P., Lubana, E","venue":"arXiv (Cornell University)","work_id":"da63c249-19b0-4d11-94bf-033eeaa2d43a","year":2024},"citing_paper":{"arxiv_id":"2606.22237","last_updated":"2026-06-20T21:37:12Z","snapshot_observed_at":"2026-08-06T06:38:36.041719Z","submitted_at":"2026-06-20T21:37:12Z","title":"Investigating The Security of Modern AI and Cloud Infrastructure","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T11:31:39.910784Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2606.22237"},"observation_digest":"sha256:64e813ad478897f383e682ed09831158505f4f7fc726bb877bf64565d39e376b","observation_id":"e2cc9341-e13b-4344-aebc-f97c301d251d","resolution":{"observed_at":"2026-07-04T08:29:42.338587Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.209911+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-07-11T22:40:37.839133Z","title":"Foundational challenges in assuring alignment and safety of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.03968","last_updated":"2026-07-09T20:41:05Z","snapshot_observed_at":"2026-08-08T01:19:08.383401Z","submitted_at":"2026-07-04T17:57:05Z","title":"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-11T22:40:37.839133Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2607.03968"},"observation_digest":"sha256:297e67aaf3c19bab51758cdb12d85b6ea7db16e5a88ba1e2a5e8f0cf7723975f","observation_id":"2811f24f-910a-4c3e-b60d-9ea0d54caf92","resolution":{"observed_at":"2026-07-11T22:40:37.839133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-07-13T07:01:49.222325Z","title":"Foundational challenges in assuring alignment and safety of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.03968","last_updated":"2026-07-09T20:41:05Z","snapshot_observed_at":"2026-08-08T01:19:08.383401Z","submitted_at":"2026-07-04T17:57:05Z","title":"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-13T07:01:49.222325Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2607.03968"},"observation_digest":"sha256:f006dc66edf419cb022e296a8b25d03616902f6ad28a197ea9e9388331b557f3","observation_id":"580bb6d7-6b73-421a-b14f-02db6794d56d","resolution":{"observed_at":"2026-07-13T07:01:49.222325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-07-31T12:20:07.096313Z","title":"Foundational","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28282","last_updated":"2026-07-30T14:31:07Z","snapshot_observed_at":"2026-08-06T16:34:19.905250Z","submitted_at":"2026-07-30T14:31:07Z","title":"(Towards) Scalable Reliable Automated Evaluation with Large Language Models","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-07-31T12:20:07.096313Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2607.28282"},"observation_digest":"sha256:93baabd36551544cf77cf0729ab63772bb8c3533cc1537f326cf4835a27ff68c","observation_id":"078c6946-2dbb-4e66-bc77-84bc0ee45a14","resolution":{"observed_at":"2026-07-31T12:20:07.096313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-10T04:29:11.241994Z","title":"Foundational challenges in assuring alignment and safety of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.07446","last_updated":"2026-08-07T17:33:09Z","snapshot_observed_at":"2026-08-11T13:13:39.227150Z","submitted_at":"2026-08-07T17:33:09Z","title":"Taxonomy-Driven Analysis of Open-Source AI Risk Mitigation Tools","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T04:29:11.241994Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2608.07446"},"observation_digest":"sha256:bd8e11eae8cf31bae4a108d56e722e5462bcfc55e9280c05bc45d4ff123a15af","observation_id":"b30a7b2d-a07f-4801-bbb5-93ca6fd4e08b","resolution":{"observed_at":"2026-08-10T04:29:11.241994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2404.09932/citation-record","integrity":"/paper/2404.09932/integrity","json":"/paper/2404.09932/citation-record.json","paper":"/paper/2404.09932"},"outbound":[],"paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T18:00:30.424554Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 57 inbound Pith citation observations for arXiv:2404.09932."}