{"as_of":"2026-08-11T03:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6faf8fc8ced1be531fd30a7fc836e14a69c0ab667b3d4102b9f2dbab815b6443","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":46,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":46,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:54:02.030313Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":18,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2402.19173","last_updated":"2024-02-29T13:53:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-29T13:53:35Z","title":"StarCoder 2 and The Stack v2: The Next Generation","version":1},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-05-12T17:28:22.353355Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2402.19173"},"observation_digest":"sha256:f75557cdd519767d7a00bd63e0eb7674816a6bb6af0700b4f155de1050d50578","observation_id":"d71ed716-aa49-42ae-a6c4-491f44c60ad0","resolution":{"observed_at":"2026-05-12T17:28:22.675474Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2411.10656","last_updated":"2026-04-03T16:24:12Z","snapshot_observed_at":"2026-07-06T19:51:14.080262Z","submitted_at":"2024-11-16T01:31:29Z","title":"Precision or Peril: A PoC of Python Code Quality from Quantized Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T17:30:54.204300Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2411.10656"},"observation_digest":"sha256:65269f905b56ad8bd288b9d7d4068cddaa22160d4592729650d2064e75ccb95d","observation_id":"4299a779-96e4-4419-9105-def9b99382a0","resolution":{"observed_at":"2026-05-23T17:33:14.955539Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-10T20:54:02.030313Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07124","last_updated":"2025-01-17T09:39:17Z","snapshot_observed_at":"2026-08-10T22:14:07.704280Z","submitted_at":"2025-01-13T08:26:43Z","title":"LLM360 K2: Building a 65B 360-Open-Source Large Language Model from Scratch","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T20:54:02.030313Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2501.07124"},"observation_digest":"sha256:2dc025c22fc1e706cf8e9b6195a46d511cdf08bfad1a40c6708466751d5042d0","observation_id":"34a0f1f4-db22-4fa9-a52c-942427e07f4a","resolution":{"observed_at":"2026-08-10T20:54:02.030313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-10T20:33:32.868576Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08200","last_updated":"2025-01-14T15:27:01Z","snapshot_observed_at":"2026-08-10T20:27:08.514445Z","submitted_at":"2025-01-14T15:27:01Z","title":"CWEval: Outcome-driven Evaluation on Functionality and Security of LLM Code Generation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:32.868576Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2501.08200"},"observation_digest":"sha256:813bee36d6a105ad2ff6251b770ac9b8a58ced8ec3611651855390b562dad6cb","observation_id":"8ff9165e-9e70-4832-be1e-67814bb5b3d5","resolution":{"observed_at":"2026-08-10T20:33:32.868576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:40:50.139345Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2501.14249"},"observation_digest":"sha256:0d5d9fa018cfc3b04d4403d1ff52e513afd00505b428e899e45f18b05f196785","observation_id":"f303afb9-da19-4460-9313-d870f3e455bd","resolution":{"observed_at":"2026-05-10T18:40:50.368040Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-09T13:45:55.122362Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.02009","last_updated":"2025-02-04T04:56:34Z","snapshot_observed_at":"2026-08-09T23:28:28.808011Z","submitted_at":"2025-02-04T04:56:34Z","title":"LLMSecConfig: An LLM-Based Approach for Fixing Software Container Misconfigurations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-09T13:45:55.122362Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2502.02009"},"observation_digest":"sha256:4ca32c152bdbd916b93666851ffd488881cfab742bbe82412ef7a885bfd9c73d","observation_id":"301c7b7c-519f-472b-9145-8a07f717f1db","resolution":{"observed_at":"2026-08-09T13:45:55.122362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-08T17:00:30.464049Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06039","last_updated":"2025-02-09T21:23:07Z","snapshot_observed_at":"2026-08-09T23:27:31.623998Z","submitted_at":"2025-02-09T21:23:07Z","title":"Benchmarking Prompt Engineering Techniques for Secure Code Generation with GPT Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T17:00:30.464049Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2502.06039"},"observation_digest":"sha256:269201e684a00c39284ee94593bb26dff80b3377dbb7fe095616c5a5b3a11541","observation_id":"3e89f275-13dc-49a2-8358-52d8b34ab118","resolution":{"observed_at":"2026-08-08T17:00:30.464049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T13:04:58.231629Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22704","last_updated":"2025-05-28T17:57:47Z","snapshot_observed_at":"2026-08-09T22:36:48.949951Z","submitted_at":"2025-05-28T17:57:47Z","title":"Training Language Models to Generate Quality Code with Program Analysis Feedback","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T13:04:58.231629Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2505.22704"},"observation_digest":"sha256:b029aa08282a961f9a9e7a6c63e05d855ef5fef0b8372b8dc54fdad6503b2967","observation_id":"259fba11-9ee5-4183-a195-facc5c07f03b","resolution":{"observed_at":"2026-08-07T13:04:58.231629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T12:46:10.267103Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23634","last_updated":"2025-05-29T16:44:29Z","snapshot_observed_at":"2026-08-09T15:33:48.720032Z","submitted_at":"2025-05-29T16:44:29Z","title":"MCP Safety Training: Learning to Refuse Falsely Benign MCP Exploits using Improved Preference Alignment","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:46:10.267103Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2505.23634"},"observation_digest":"sha256:abe6e3fe277be37ce9b8b40102ed2a6c2a717d29371c8cd5940efa54c7791e04","observation_id":"6f606523-bef9-46e5-b411-10bf0bb1bb23","resolution":{"observed_at":"2026-08-07T12:46:10.267103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T11:52:57.988743Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02066","last_updated":"2025-06-01T23:37:41Z","snapshot_observed_at":"2026-08-09T07:08:16.872886Z","submitted_at":"2025-06-01T23:37:41Z","title":"Developing a Risk Identification Framework for Foundation Model Uses","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:57.988743Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.02066"},"observation_digest":"sha256:6250e412fe090115c8439aaa6c60be37894287abe1b54efb797c558c39821135","observation_id":"2a740e82-3223-4dfb-9470-ccff15569fbc","resolution":{"observed_at":"2026-08-07T11:52:57.988743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T05:42:54.911928Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07313","last_updated":"2025-06-08T23:08:08Z","snapshot_observed_at":"2026-08-10T11:09:11.672263Z","submitted_at":"2025-06-08T23:08:08Z","title":"SCGAgent: Recreating the Benefits of Reasoning Models for Secure Code Generation with Agentic Workflows","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:42:54.911928Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.07313"},"observation_digest":"sha256:cccece429e5b32a67501b12f018c4ea9853214a37d958256d5ec97af1cae5f96","observation_id":"9e5f17a6-a27d-47b9-b7eb-835e73519e26","resolution":{"observed_at":"2026-08-07T05:42:54.911928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T10:17:26.763796Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.763796Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:bd2d60810ccf317f2e07878a04b815df34fa2a3cad29532abb5b9574bf09f69d","observation_id":"78188daa-cfd8-411e-b556-0bf09ef0c4f5","resolution":{"observed_at":"2026-08-07T10:17:26.763796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T21:56:13.509418Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23034","last_updated":"2025-06-28T23:24:33Z","snapshot_observed_at":"2026-08-08T22:22:43.604466Z","submitted_at":"2025-06-28T23:24:33Z","title":"Guiding AI to Fix Its Own Flaws: An Empirical Study on LLM-Driven Secure Code Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:56:13.509418Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.23034"},"observation_digest":"sha256:6235eb12aa0d933f5303c63a1ecfa82248ae891d1e5ee1686725fa40258899e4","observation_id":"381c30dd-e45d-4868-af24-85cbb8ecd9f9","resolution":{"observed_at":"2026-08-06T21:56:13.509418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T20:45:53.247909Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02057","last_updated":"2025-07-02T18:00:49Z","snapshot_observed_at":"2026-08-10T11:19:13.970097Z","submitted_at":"2025-07-02T18:00:49Z","title":"MGC: A Compiler Framework Exploiting Compositional Blindness in Aligned LLMs for Malware Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:45:53.247909Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2507.02057"},"observation_digest":"sha256:4699e618ac6712cd08c47b1b6a3442c02aa7424bc403a6ef4870ef7265df94c8","observation_id":"cf6837e0-a1cf-456c-b41d-df3c068a6907","resolution":{"observed_at":"2026-08-06T20:45:53.247909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T14:45:42.847281Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models.arXiv preprint arXiv:2312.04724,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.18105","last_updated":"2025-07-24T05:30:54Z","snapshot_observed_at":"2026-08-07T21:22:02.859315Z","submitted_at":"2025-07-24T05:30:54Z","title":"Understanding the Supply Chain and Risks of Large Language Model Applications","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:42.847281Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2507.18105"},"observation_digest":"sha256:788fbd8e77d83ab265d9884f6bece67a7b7cfbda183e266b13d3c89c2d7c9208","observation_id":"97794215-ada6-49f4-9f3e-40a02bb0b665","resolution":{"observed_at":"2026-08-06T14:45:42.847281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T14:25:15.757169Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19399","last_updated":"2025-07-25T16:06:16Z","snapshot_observed_at":"2026-08-09T23:41:06.417558Z","submitted_at":"2025-07-25T16:06:16Z","title":"Running in CIRCLE? A Simple Benchmark for LLM Code Interpreter Security","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T14:25:15.757169Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2507.19399"},"observation_digest":"sha256:34669ef93ac5769a6c256e203fb23c06399eadb44b794642dd0fd62d5ee76fca","observation_id":"5dfe7ef9-3eb8-4fa5-8515-d47d7e71faec","resolution":{"observed_at":"2026-08-06T14:25:15.757169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T01:03:49.483750Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.03933","last_updated":"2025-08-05T21:55:14Z","snapshot_observed_at":"2026-08-10T07:07:18.397746Z","submitted_at":"2025-08-05T21:55:14Z","title":"Towards terahertz nanomechanics","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T01:03:49.483750Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2508.03933"},"observation_digest":"sha256:6aef0d16d0d4033c0d4e2e3eb7b2c89e00f758a6868bb43b2b879fa614505293","observation_id":"24a861b7-a4e2-4ef7-8c4e-c1211673c999","resolution":{"observed_at":"2026-08-06T01:03:49.483750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T01:04:32.435944Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.03936","last_updated":"2025-08-05T21:57:52Z","snapshot_observed_at":"2026-08-08T12:25:04.397522Z","submitted_at":"2025-08-05T21:57:52Z","title":"ASTRA: Autonomous Spatial-Temporal Red-teaming for AI Software Assistants","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T01:04:32.435944Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2508.03936"},"observation_digest":"sha256:45e7dd297c20b2810eb7291e8a0c0160cfaa08f9314d2f2f16c34b36572ee66d","observation_id":"75bcb548-44e5-45d4-8578-254e6d951780","resolution":{"observed_at":"2026-08-06T01:04:32.435944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-03T23:50:49.056328Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.03898","last_updated":"2025-11-05T22:46:24Z","snapshot_observed_at":"2026-08-07T13:07:19.220144Z","submitted_at":"2025-11-05T22:46:24Z","title":"Secure Code Generation at Scale with Reflexion","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T23:50:49.056328Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2511.03898"},"observation_digest":"sha256:58f5b6745cadab884e7cb6e5556a0e03eb1cc259975bf72e6b4a576dcd374055","observation_id":"1f80b99b-ddea-4efa-bee6-62a473fe1d3f","resolution":{"observed_at":"2026-08-03T23:50:49.056328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2512.05439","last_updated":"2026-05-07T21:46:58Z","snapshot_observed_at":"2026-08-06T07:24:27.198375Z","submitted_at":"2025-12-05T05:34:06Z","title":"BEAVER: An Efficient Deterministic LLM Verifier","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T01:58:44.719715Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2512.05439"},"observation_digest":"sha256:d9383ce296f8449e1a2c52a78936a6aefa8e4d038de8ec0003c68750a6569d85","observation_id":"375b5ec1-aada-443f-8522-c7944673692a","resolution":{"observed_at":"2026-05-17T01:58:51.237959Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-03T05:18:42.953661Z","title":"Cao, X., Jia, J., and Gong, N","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.04894","last_updated":"2026-06-05T16:30:46Z","snapshot_observed_at":"2026-08-06T23:30:53.886306Z","submitted_at":"2026-02-02T22:23:36Z","title":"Extracting Recurring Vulnerabilities from Black-Box LLM-Generated Software","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T05:18:42.953661Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2602.04894"},"observation_digest":"sha256:3f967eff9ff7f96a268d5850a7bd9d7ff63dadf02b151f8c7316e05ddb6a1232","observation_id":"12076545-9314-47ab-8e36-96d721f8ed1e","resolution":{"observed_at":"2026-08-03T05:18:42.953661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2602.06759","last_updated":"2026-05-14T07:04:32Z","snapshot_observed_at":"2026-08-04T02:39:11.494645Z","submitted_at":"2026-02-06T15:06:36Z","title":"\"Tab, Tab, Bug\": Security Pitfalls of Next Edit Suggestions in AI-Integrated IDEs","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T06:53:13.235588Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2602.06759"},"observation_digest":"sha256:f973cc55a49edb61fd0e45f07f533810abd57be759e350d2f602b2634c4be424","observation_id":"1c6d0de2-b199-4885-869e-e2348c008998","resolution":{"observed_at":"2026-05-16T06:57:29.297748Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.05292","last_updated":"2026-04-08T16:49:43Z","snapshot_observed_at":"2026-07-06T22:54:04.491386Z","submitted_at":"2026-04-07T00:55:42Z","title":"Broken by Default: A Formal Verification Study of Security Vulnerabilities in AI-Generated Code","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T20:12:03.118074Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.05292"},"observation_digest":"sha256:05388e3e77da52c5e04f3f0ecf38e0d179408141dd383d9ded5375b1b62f9e16","observation_id":"94d15de5-8ac2-4d03-9fd8-6d3157135cbb","resolution":{"observed_at":"2026-05-10T22:10:48.566036Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.09544","last_updated":"2026-07-03T15:37:04Z","snapshot_observed_at":"2026-08-01T18:28:17.308608Z","submitted_at":"2026-04-10T17:58:31Z","title":"Large Language Models Generate Harmful Responses Using a Distinct Mechanism, Shared Across Harm Types","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T17:08:25.471462Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.09544"},"observation_digest":"sha256:6349233af4994b6c1c9e141947ac276cac61cde3658b3afa18d5e97028dc37fc","observation_id":"bebf90b3-ee10-42c9-9deb-daf54a51f640","resolution":{"observed_at":"2026-05-11T07:31:00.338647Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.17803","last_updated":"2026-04-20T04:51:39Z","snapshot_observed_at":"2026-08-02T12:38:39.791138Z","submitted_at":"2026-04-20T04:51:39Z","title":"Adversarial Arena: Crowdsourcing Data Generation through Interactive Competition","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-10T04:55:43.987116Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.17803"},"observation_digest":"sha256:ca72a4983aa4b9aad0656107543220e8cc34a85c7bb40289eb9a6112b6b1e8b9","observation_id":"7fbbb1e4-aa8e-49f4-b162-b551afa5aa3e","resolution":{"observed_at":"2026-05-10T11:10:09.437088Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.18718","last_updated":"2026-04-20T18:17:51Z","snapshot_observed_at":"2026-08-05T14:04:49.062431Z","submitted_at":"2026-04-20T18:17:51Z","title":"Towards Optimal Agentic Architectures for Offensive Security Tasks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T04:02:04.359269Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.18718"},"observation_digest":"sha256:5a58b1d81480ea8d08966018e73fe84a117a581fa6d72a926101eb851b792123","observation_id":"2c5d9f52-8c89-4d6e-9660-35e553710ce1","resolution":{"observed_at":"2026-05-11T12:16:03.146927Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.03179","last_updated":"2026-05-04T21:42:10Z","snapshot_observed_at":"2026-08-05T14:29:07.431614Z","submitted_at":"2026-05-04T21:42:10Z","title":"A Validated Prompt Bank for Malicious Code Generation: Separating Executable Weapons from Security Knowledge in 1,554 Consensus-Labeled Prompts","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-08T18:11:29.066362Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.03179"},"observation_digest":"sha256:b3bdee7caf0c300c62bfa634d1a6444ccf1e9406a9c4351c23fb907077c7ede7","observation_id":"3eda2f9c-237c-4b36-803e-1bdc8981c0d5","resolution":{"observed_at":"2026-05-09T06:40:43.797249Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.04019","last_updated":"2026-05-05T17:43:52Z","snapshot_observed_at":"2026-07-06T23:16:49.553583Z","submitted_at":"2026-05-05T17:43:52Z","title":"Redefining AI Red Teaming in the Agentic Era: From Weeks to Hours","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-07T16:06:18.057868Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.04019"},"observation_digest":"sha256:07214e1718fede16f6f34df96a4039a354fbdb0fe0014312a9a12c7468eb55ed","observation_id":"89af28ed-bcf5-40d3-8611-5630b43ad2ef","resolution":{"observed_at":"2026-05-11T23:51:44.776533Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.05267","last_updated":"2026-05-06T09:38:31Z","snapshot_observed_at":"2026-08-04T17:41:12.164736Z","submitted_at":"2026-05-06T09:38:31Z","title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-08T17:37:51.790000Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.05267"},"observation_digest":"sha256:3c1edcc4f870afd51ce74d8b38f019c01299d80073a3447381e3ddf1ba4a11b5","observation_id":"15bad8b2-2d34-44a3-928e-b4a531ea6e94","resolution":{"observed_at":"2026-05-11T17:26:04.477648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.08382","last_updated":"2026-05-08T18:40:47Z","snapshot_observed_at":"2026-08-11T02:43:57.873082Z","submitted_at":"2026-05-08T18:40:47Z","title":"SecureForge: Finding and Preventing Vulnerabilities in LLM-Generated Code via Prompt Optimization","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T01:09:08.378040Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.08382"},"observation_digest":"sha256:b44575016771c8c4d20525d97e534fef5d4dfe6774f8b6540a718a309b5d0179","observation_id":"593fda4e-e88b-4287-a655-07e4b936771f","resolution":{"observed_at":"2026-05-12T08:26:24.588493Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.08898","last_updated":"2026-05-09T11:43:47Z","snapshot_observed_at":"2026-07-06T23:21:02.177557Z","submitted_at":"2026-05-09T11:43:47Z","title":"LLM-Agnostic Semantic Representation Attack","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-12T01:14:08.629862Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.08898"},"observation_digest":"sha256:ddac151f8e5aa90de04ecba1f53c2b9fff5d5f8ba925402f3f9d01aff01c1b29","observation_id":"d2708aee-a774-4683-95be-2aeb7caa45d7","resolution":{"observed_at":"2026-05-12T08:21:24.222423Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.17413","last_updated":"2026-05-17T12:18:20Z","snapshot_observed_at":"2026-08-03T01:59:45.581393Z","submitted_at":"2026-05-17T12:18:20Z","title":"Ablating Safety: Mechanisms for Removing Alignment in Language Models for Security Applications","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T23:30:43.364230Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.17413"},"observation_digest":"sha256:8531980c983df94c71ff61d7e4f8130e946e51888fff7798044280878e1e6c6b","observation_id":"de0b055b-e787-4943-bf8c-d0ea299bb904","resolution":{"observed_at":"2026-05-19T23:32:52.541609Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.20351","last_updated":"2026-05-19T18:05:51Z","snapshot_observed_at":"2026-07-06T23:30:58.549353Z","submitted_at":"2026-05-19T18:05:51Z","title":"Refusal Evaluation in Coding LLMs and Code Agents: A Systematic Review of Thirteen Malicious-Code Prompt Corpora (2023-2025)","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:43.363709Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.20351"},"observation_digest":"sha256:f0854bbb7a773a54612a54077962f7352688a3a064682e661010ceabad9c9728","observation_id":"e9453341-ea23-434c-9bd0-857ce0aca0b2","resolution":{"observed_at":"2026-05-21T07:39:48.536733Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.21773","last_updated":"2026-05-20T22:07:12Z","snapshot_observed_at":"2026-07-06T23:32:10.472558Z","submitted_at":"2026-05-20T22:07:12Z","title":"HIDBench: Benchmarking Large Language Models for Host-Based Intrusion Detection","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T08:52:34.079804Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.21773"},"observation_digest":"sha256:16ef156f680b3e71df6c35c7bd709f96351e1997f57c16abc82c6216e654b566","observation_id":"014b4ae9-6471-42c6-9e79-be13ffc29e50","resolution":{"observed_at":"2026-05-22T08:54:45.868682Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T05:50:28.114140Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:71722c08218fc6131ed4c4e1df5eb57e71714c7e453e1b7c5eb8516ba9a4522d","observation_id":"07a96601-9bbd-4e36-bdfd-14c9a7bae55b","resolution":{"observed_at":"2026-05-22T05:51:07.706475Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-25T06:05:27.736494Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:d456b5f17ce304d05f6eea1257b79be2b502e362d28bb201099c73739b7f9bd0","observation_id":"50142792-7fa7-41bc-bfa7-d3d3e189d660","resolution":{"observed_at":"2026-05-25T06:06:43.062405Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.23091","last_updated":"2026-05-21T22:53:40Z","snapshot_observed_at":"2026-08-01T20:58:37.779697Z","submitted_at":"2026-05-21T22:53:40Z","title":"Security of LLM-generated Code: A Comparative Analysis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T05:16:26.372764Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.23091"},"observation_digest":"sha256:e7828cb07514c7d383b2b9ca103112c8c732db3173d83916f74ff4d7d8283b6e","observation_id":"8497cf12-ab6e-4c82-9ac5-566da15afb6a","resolution":{"observed_at":"2026-05-25T05:16:39.185029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2606.01317","last_updated":"2026-05-31T16:06:02Z","snapshot_observed_at":"2026-08-10T05:39:13.866749Z","submitted_at":"2026-05-31T16:06:02Z","title":"SABER: Benchmarking Operational Safety of LLM Coding Agents in Stateful Project Workspaces","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-28T16:44:18.994680Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2606.01317"},"observation_digest":"sha256:3bb702286afcc9d97f735dccc7273d78456e72447332b874434a7f93088701de","observation_id":"0e47e741-ab83-4a5a-aa3c-4c83b18a6e4d","resolution":{"observed_at":"2026-07-01T21:36:15.066149Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2606.25973","last_updated":"2026-06-24T15:45:38Z","snapshot_observed_at":"2026-08-07T03:45:14.020128Z","submitted_at":"2026-06-24T15:45:38Z","title":"Helpful or Harmful? Evaluating LLM-Assisted Vulnerability Patching via a Human Study","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-25T19:02:45.109478Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2606.25973"},"observation_digest":"sha256:665cabc8d5e799d47c65d1f58273e6d83ceb3feb8815c404984b94a5e91329fd","observation_id":"e7a4f50f-7229-44a5-a12a-25664d9e796a","resolution":{"observed_at":"2026-07-04T21:10:09.288469Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2606.29175","last_updated":"2026-06-28T03:43:30Z","snapshot_observed_at":"2026-07-07T00:03:12.330608Z","submitted_at":"2026-06-28T03:43:30Z","title":"Direct Causation in International Humanitarian Law and the Challenge of AI-Mediated Civilian Cyber Operations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T07:49:28.819402Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2606.29175"},"observation_digest":"sha256:e495535bc0b1454fce0853074f7adf629c955fd9fe403d75e5242497505cce6d","observation_id":"9d048307-d19b-422f-a380-eb09b9fd25ee","resolution":{"observed_at":"2026-06-30T07:54:22.125211Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-07-12T07:33:43.015966Z","title":"PurpleLlama CyberSecEval: A secure coding benchmark for language models.arXiv preprint arXiv:2312.04724,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02714","last_updated":"2026-07-07T12:39:55Z","snapshot_observed_at":"2026-08-07T02:57:17.862073Z","submitted_at":"2026-07-02T19:05:07Z","title":"Not All Refusals Are Equal: How Safety Alignment Fails Cybersecurity at Scale","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-12T07:33:43.015966Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.02714"},"observation_digest":"sha256:ed366085032155a0ce06eae29fe9ae82ca4623ca19354989d8a27dba5e050075","observation_id":"f2caab62-2233-4b1d-a14f-600eca1074a7","resolution":{"observed_at":"2026-07-12T07:33:43.015966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-07-11T22:40:37.839133Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03968","last_updated":"2026-07-09T20:41:05Z","snapshot_observed_at":"2026-08-08T01:19:08.383401Z","submitted_at":"2026-07-04T17:57:05Z","title":"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-11T22:40:37.839133Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.03968"},"observation_digest":"sha256:0f2044c74613e4e037634a94208f58695299712d79dece341c249ed27d893973","observation_id":"0d6179b5-8f42-4710-af49-7b22b97a1710","resolution":{"observed_at":"2026-07-11T22:40:37.839133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-07-13T07:01:49.222325Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03968","last_updated":"2026-07-09T20:41:05Z","snapshot_observed_at":"2026-08-08T01:19:08.383401Z","submitted_at":"2026-07-04T17:57:05Z","title":"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-13T07:01:49.222325Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.03968"},"observation_digest":"sha256:1235cb3949560dbff0208d25432342820b0d06ec16091a33e4f882f10e430ec5","observation_id":"c4b6b2a4-983a-46fb-a945-a3a224793c59","resolution":{"observed_at":"2026-07-13T07:01:49.222325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-01T02:40:29.099157Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25379","last_updated":"2026-08-01T10:03:47Z","snapshot_observed_at":"2026-08-06T23:11:27.985138Z","submitted_at":"2026-07-28T07:34:37Z","title":"Cyber-Capable AI Agents: Vulnerabilities, Evaluation Containment, and Defensive Response","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T02:40:29.099157Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.25379"},"observation_digest":"sha256:ea1c58267f4469ec0b19bf19aa99e777ab9bc52313f65ec0f7e5d1ebcb808037","observation_id":"237e36ec-cebd-45cf-9c43-8a1d2b584881","resolution":{"observed_at":"2026-08-01T02:40:29.099157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-04T01:33:17.643282Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25379","last_updated":"2026-08-01T10:03:47Z","snapshot_observed_at":"2026-08-06T23:11:27.985138Z","submitted_at":"2026-07-28T07:34:37Z","title":"Cyber-Capable AI Agents: Vulnerabilities, Evaluation Containment, and Defensive Response","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T01:33:17.643282Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.25379"},"observation_digest":"sha256:08adacb5fac2b39e450b61838cc9ae2075d8e5725bfeb6a240b7f7ab6aae0906","observation_id":"c2eb78e9-4e54-4e10-a6a0-07da0442523b","resolution":{"observed_at":"2026-08-04T01:33:17.643282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-08T19:58:17.193944Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models.arXiv preprint arXiv:2312.04724, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04317","last_updated":"2026-08-05T00:54:57Z","snapshot_observed_at":"2026-08-09T19:45:40.565565Z","submitted_at":"2026-08-05T00:54:57Z","title":"Trident : How to Break Deep Reinforcement Learning Cyber Defenses (Agentic)","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T19:58:17.193944Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2608.04317"},"observation_digest":"sha256:432203cac43574edea26fe90a4558541557bd9f49936f1bffa2cdc5576f23a14","observation_id":"7cdc2bd7-d8d6-4bb1-8996-da6ea68649a7","resolution":{"observed_at":"2026-08-08T19:58:17.193944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2312.04724/citation-record","integrity":"/paper/2312.04724/integrity","json":"/paper/2312.04724/citation-record.json","paper":"/paper/2312.04724"},"outbound":[],"paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-10T13:10:29.115486Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 46 inbound Pith citation observations for arXiv:2312.04724."}