{"as_of":"2026-08-09T22:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:32e7aab35a7ce0d1a25bcc2c07c1865037b69df00ad080bd3714cd116977a509","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-31T16:06:06.687929Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.24392/citation-record","integrity":"/paper/2607.24392/integrity","json":"/paper/2607.24392/citation-record.json","paper":"/paper/2607.24392"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-07-31T16:06:03.615181Z","title":"Awadalla","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:03.615181Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:14cfd104887bbc399701eca27489be53c56d12be1601f241ab6f10c937981034","observation_id":"b310d6db-cb48-4ee2-bb07-75581854612d","resolution":{"observed_at":"2026-07-31T16:06:03.615181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-07-31T16:06:03.689934Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:03.689934Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:a56f81bb41904ec2c149a1e7a5fcedc2bc4a6a8bdcf2a5515953bfa86b4dbd8c","observation_id":"a0f0e8e9-48cd-486a-8129-9dd083cabf0f","resolution":{"observed_at":"2026-07-31T16:06:03.689934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:03.783409Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:03.783409Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:986732c3470969873fdd80771a877f829a55963b6ebb6357a12264f9d5bab745","observation_id":"b509488a-218b-4b99-83aa-3e9b2ac83105","resolution":{"observed_at":"2026-07-31T16:06:03.783409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14132","last_updated":"2023-11-07T03:30:15Z","snapshot_observed_at":"2026-07-06T16:10:54.723336Z","submitted_at":"2023-08-27T15:20:06Z","title":"Detecting Language Model Attacks with Perplexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14132","snapshot_observed_at":"2026-07-31T16:06:03.882350Z","title":"Detecting language model attacks with perplexity, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:03.882350Z"},"links":{"cited_paper":"/paper/2308.14132","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:9974e461fce5383d4199328317062a4a073b199994e8b255be352f12dbab862f","observation_id":"f41706c2-33b6-48e5-86df-aece5f6d7ede","resolution":{"observed_at":"2026-07-31T16:06:03.882350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-07-31T16:06:04.089010Z","title":"Language models are few-shot learners.arXiv preprint arXiv:2005.14165, 2020","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.089010Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:59cc55cd0e77687da87374acc43f0d8a57d3f041bc3135626b8940b77a003d62","observation_id":"aeeedf14-617a-4aed-8c71-cef91edcac68","resolution":{"observed_at":"2026-07-31T16:06:04.089010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-31T16:06:04.193653Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.193653Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:9a11d096fa1f52dd1ea57618acea84ff77d6efc28f11b75382e978984b16ed49","observation_id":"253a77cf-dfb4-4256-b0df-ebfab08a46fc","resolution":{"observed_at":"2026-07-31T16:06:04.193653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-07-06T18:23:23.565502Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-07-31T16:06:04.281114Z","title":"Or-bench: An over-refusal benchmark for large language models.arXiv preprint arXiv:2405.20947, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.281114Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:4c0fec863fab83ac688712fc5ca93c6a1c28de57849c30ee13265c91953d0ea9","observation_id":"75a6e83e-7179-488d-8857-cf1a5ded795f","resolution":{"observed_at":"2026-07-31T16:06:04.281114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:04.364227Z","title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.364227Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:72b7bfb2dcf4faf33644d1ea61aa71f3fd6e38cb02590cb53f8a378df60e47b4","observation_id":"15352712-e2f2-4e54-9f8d-cb9f642e26d9","resolution":{"observed_at":"2026-07-31T16:06:04.364227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:04.511164Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.511164Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:b265910bf358fcfb6ca2f5362bf442203d99fe1b7de3cd310df16ac01bad8350","observation_id":"7f055f78-ed1d-47af-9e52-cff5c7013f2b","resolution":{"observed_at":"2026-07-31T16:06:04.511164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16379","last_updated":"2024-06-21T07:35:53Z","snapshot_observed_at":"2026-07-06T17:35:23.161066Z","submitted_at":"2024-02-26T07:58:12Z","title":"TEaR: Improving LLM-based Machine Translation with Systematic Self-Refinement","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16379","snapshot_observed_at":"2026-07-31T16:06:04.707133Z","title":"Improving llm-based machine translation with systematic self-correction.arXiv preprint arXiv:2402.16379, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.707133Z"},"links":{"cited_paper":"/paper/2402.16379","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:eb4b11a8593aafc7d1961542bbc15b78087a89b98221da1199f70347b6de658c","observation_id":"88a8a4ce-aa82-44e0-91ed-98d7ce5e593c","resolution":{"observed_at":"2026-07-31T16:06:04.707133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:04.784376Z","title":"The robots are coming: Exploring the implications of openai codex on introductory program- ming","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.784376Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:2d08e643252efc87284ada0985195ee2434effb95b0031b4ba188389509499a3","observation_id":"f69015bc-e1bd-4c3f-a985-fdb2956ef072","resolution":{"observed_at":"2026-07-31T16:06:04.784376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06674","last_updated":"2023-12-07T19:40:50Z","snapshot_observed_at":"2026-07-06T17:00:00.321552Z","submitted_at":"2023-12-07T19:40:50Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06674","snapshot_observed_at":"2026-07-31T16:06:04.876439Z","title":"Upasani, Jianfeng Chi, Rashi Rungta, Krithika Iyer, Yuning Mao, Michael Tontchev, Qing Hu, Brian Fuller, Davide Testuggine, and Madian Khabsa","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.876439Z"},"links":{"cited_paper":"/paper/2312.06674","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:87ba3d97affde2c58fd7f1cf8d6394df229a169c3912f08a17ff77bbc49bea99","observation_id":"ff5872d1-564f-43b3-995c-f00c0877dac7","resolution":{"observed_at":"2026-07-31T16:06:04.876439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00614","last_updated":"2023-09-04T17:47:36Z","snapshot_observed_at":"2026-07-06T16:13:23.343694Z","submitted_at":"2023-09-01T17:59:44Z","title":"Baseline Defenses for Adversarial Attacks Against Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00614","snapshot_observed_at":"2026-07-31T16:06:04.944283Z","title":"Baseline defenses for adversarial attacks against aligned language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.944283Z"},"links":{"cited_paper":"/paper/2309.00614","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:73cad62a2b2b81492afc58de961f571550ef43108362ca436810d2bbc8d00c93","observation_id":"0dab7cd4-8336-406d-9b0c-f54cbe7c67a6","resolution":{"observed_at":"2026-07-31T16:06:04.944283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-07-31T16:06:05.004532Z","title":"Jiang, Alexandre Sablayrolles, and Arthur et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.004532Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:c78687bd11c1c2d4434db3e1cf1ad27dae08fa1adf955964c9c803eaa9bc7e8c","observation_id":"5c447e52-6f8a-43f2-878c-c4f64c0f9dea","resolution":{"observed_at":"2026-07-31T16:06:05.004532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:05.089361Z","title":"Openassistant conversations-democratizing large language model alignment.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.089361Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:b32a3849c054b957c1f5ef254801e48df2a96c0ebe68f618f8158ed99dce0128","observation_id":"0accd163-a524-4631-a378-50070b71ae7e","resolution":{"observed_at":"2026-07-31T16:06:05.089361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:05.156920Z","title":"Salad-bench: A hierarchical and comprehensive safety benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.156920Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:ffd60ea062f0c73d48381a5d57caa770e0f21f97bce971c2b2c9fcf53d51e1df","observation_id":"b1887b75-0793-4ee5-b58c-1ef94547d7d0","resolution":{"observed_at":"2026-07-31T16:06:05.156920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.13540","last_updated":"2024-12-12T11:00:40Z","snapshot_observed_at":"2026-08-06T08:11:31.274837Z","submitted_at":"2024-07-18T14:15:46Z","title":"Counting function estimates for coherent frames and Riesz sequences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.13540","snapshot_observed_at":"2026-07-31T16:06:05.219804Z","title":"You can’t eat your cake and have it too: The performance degradation of llms with jailbreak defense.arXiv preprint arXiv:2407.13540, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.219804Z"},"links":{"cited_paper":"/paper/2407.13540","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:531bc0e2c45cd82cab95ac0aef59d4beb815e7727bc1c953c3260c8b67d91f26","observation_id":"195d7399-377a-4024-b7bc-786eb3d8fdc8","resolution":{"observed_at":"2026-07-31T16:06:05.219804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:05.279438Z","title":"Llm self defense: By self examination, llms know they are being tricked,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.279438Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:ec20c3dfe593b4208a59c8ce65c4af129ff5a91ba800f2d3f6c8421fb88ac5bc","observation_id":"1486b3ba-5640-47a6-b238-8ae4f0b015bf","resolution":{"observed_at":"2026-07-31T16:06:05.279438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03684","last_updated":"2024-06-11T19:02:52Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:01:53Z","title":"SmoothLLM: Defending Large Language Models Against Jailbreaking Attacks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03684","snapshot_observed_at":"2026-07-31T16:06:05.405942Z","title":"Smoothllm: Defending large language models against jailbreaking attacks.arXiv preprint arXiv:2310.03684, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.405942Z"},"links":{"cited_paper":"/paper/2310.03684","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:e8e9beac89befe9c846831b1f7e0e7dbee43e42516237421258562badf7140bc","observation_id":"b33c1953-7071-425f-b150-2142b1e69023","resolution":{"observed_at":"2026-07-31T16:06:05.405942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.07308","last_updated":"2024-05-02T14:28:39Z","snapshot_observed_at":"2026-08-01T14:42:09.926151Z","submitted_at":"2023-08-14T17:54:10Z","title":"LLM Self Defense: By Self Examination, LLMs Know They Are Being Tricked","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.07308","snapshot_observed_at":"2026-07-31T16:06:05.342582Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.342582Z"},"links":{"cited_paper":"/paper/2308.07308","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:a752bd4894eb5a8b6ac19d7e365dc160e56efe25d16861d5a14160455a683812","observation_id":"0789fb59-fc4e-4340-8c58-463e14c888d5","resolution":{"observed_at":"2026-07-31T16:06:05.342582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:05.528177Z","title":"Trustllm: Trustworthiness in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.528177Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:d1c81f8a2161b139c6f09c3b5dcf0afe499297997fa88914155134d2e5520b04","observation_id":"98057fa9-a562-4330-af10-eeafa3d03830","resolution":{"observed_at":"2026-07-31T16:06:05.528177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:05.465819Z","title":"XSTest: A test suite for identifying exaggerated safety behaviours in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.465819Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:6ce5da258777e562e20aab046f0ad7f1730ae12c4985ef41ebb4f3a8dd930db9","observation_id":"65dc2ad8-1853-470a-88f6-8d4714f294c0","resolution":{"observed_at":"2026-07-31T16:06:05.465819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-07-31T16:06:05.675062Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.675062Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:dcc67c61235e68184c692086c5320a9939451a35abf20e21dc4cc0703387efb9","observation_id":"f9639ddd-ddfc-4704-bd7c-72e117b06068","resolution":{"observed_at":"2026-07-31T16:06:05.675062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05286","last_updated":"2024-10-22T03:58:20Z","snapshot_observed_at":"2026-08-04T14:54:55.050781Z","submitted_at":"2024-03-08T13:10:59Z","title":"LLM4Decompile: Decompiling Binary Code with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05286","snapshot_observed_at":"2026-07-31T16:06:05.601499Z","title":"Llm4decompile: Decompiling binary code with large language models.arXiv preprint arXiv:2403.05286, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.601499Z"},"links":{"cited_paper":"/paper/2403.05286","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:ff3d67f53376b7293f396b12e4257123c1f7e2bd6f1fdc635841cfc6e98fcb17","observation_id":"a8cb5949-4aea-4280-9c85-af3750d94922","resolution":{"observed_at":"2026-07-31T16:06:05.601499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:05.805954Z","title":"Defending LLMs against jailbreaking attacks via backtranslation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.805954Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:9f48680d0aa70c267bab55a92ac417d34fcdbab7c6a92d0dd956249242527921","observation_id":"78b46bff-5621-4fd2-83c0-15d8f16652c3","resolution":{"observed_at":"2026-07-31T16:06:05.805954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:05.742114Z","title":"The art of defending: A systematic evaluation and analysis of LLM defense strategies on safety and over-defensiveness","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.742114Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:1b8f80447525798f46092477b81940277539548252492b99f6a84cb8f86f6d3e","observation_id":"d998395c-8f7a-48c4-b26f-c5c87662bce9","resolution":{"observed_at":"2026-07-31T16:06:05.742114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06387","last_updated":"2024-05-25T07:01:15Z","snapshot_observed_at":"2026-08-06T13:25:52.403871Z","submitted_at":"2023-10-10T07:50:29Z","title":"Jailbreak and Guard Aligned Language Models with Only Few In-Context Demonstrations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06387","snapshot_observed_at":"2026-07-31T16:06:05.960205Z","title":"Jailbreak and guard aligned language models with only few in-context demonstrations, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.960205Z"},"links":{"cited_paper":"/paper/2310.06387","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:00deb527849d523f4d616ec523677ff8b1cbce01f8143bbca1dae806cd5775ad","observation_id":"7a179e9a-bb3b-445a-b683-f7a1f9e23786","resolution":{"observed_at":"2026-07-31T16:06:05.960205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-07-31T16:06:05.868142Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark.arXiv preprint arXiv:2406.01574, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:05.868142Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:7ca751bd5ab5950ecf91209e6c8ca214f5dd89acf7154dbce1d12687184234eb","observation_id":"967b658a-0064-47d9-b9c8-f210621a010e","resolution":{"observed_at":"2026-07-31T16:06:05.868142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:06.082878Z","title":"Defending chatgpt against jailbreak attack via self-reminders.Nature Machine Intelligence, 5(12):1486–1496, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.082878Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:41d7876a031e0638eb0cc98e597ec45bcb17a692dbc2672e6cc33142bf52f982","observation_id":"015aee69-d2e8-49ad-ba73-34c93b04abc1","resolution":{"observed_at":"2026-07-31T16:06:06.082878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15727","last_updated":"2024-03-04T05:37:40Z","snapshot_observed_at":"2026-08-09T16:55:44.091721Z","submitted_at":"2024-02-24T05:34:43Z","title":"LLMs Can Defend Themselves Against Jailbreaking in a Practical Manner: A Vision Paper","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15727","snapshot_observed_at":"2026-07-31T16:06:06.031150Z","title":"Llms can defend themselves against jailbreaking in a practical manner: A vision paper.ArXiv, abs/2402.15727, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.031150Z"},"links":{"cited_paper":"/paper/2402.15727","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:a818792f4ae0c77944c99cf2ebb5953ce1780426f950c1e74984ea34c314887a","observation_id":"0df3be41-3ad5-47e0-9b08-a214089f302f","resolution":{"observed_at":"2026-07-31T16:06:06.031150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:06.239596Z","title":"SafeDecoding: Defending against jailbreak attacks via safety-aware decoding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.239596Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:dfb9db32fff0a38b147d93e023cf1f2ad30c5372125e0bd01273debdc91024d6","observation_id":"dcd3b31c-d92f-45fb-89ce-84553e2d4239","resolution":{"observed_at":"2026-07-31T16:06:06.239596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:06.178608Z","title":"Contrastive preference optimization: Pushing the bound- aries of LLM performance in machine translation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.178608Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:6e1ac09369c7eda5fb42f1ac52ab25423cf84f74e23592d4fbf8c3cc140141db","observation_id":"7c08e564-bb40-4baa-ada9-8baa55f1d190","resolution":{"observed_at":"2026-07-31T16:06:06.178608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-07-31T16:06:06.390610Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.390610Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:f6ed54d5a21d574ce35f19c6403e1a7405b71f750e775ac08f8f9f069f433276","observation_id":"15a06c57-1f00-44ac-ad24-2746479fd2ae","resolution":{"observed_at":"2026-07-31T16:06:06.390610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:06.323695Z","title":"A comprehensive study of jailbreak attack versus defense for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.323695Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:24072a403af2fc75b5008ffb264bb146510982458000dee62e58c3c4916474a3","observation_id":"c66ab8a0-923c-4170-922f-8fe067ecd526","resolution":{"observed_at":"2026-07-31T16:06:06.323695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06561","last_updated":"2024-12-16T08:43:24Z","snapshot_observed_at":"2026-08-09T02:31:23.206221Z","submitted_at":"2024-01-12T13:15:05Z","title":"Intention Analysis Makes LLMs A Good Jailbreak Defender","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06561","snapshot_observed_at":"2026-07-31T16:06:06.522729Z","title":"Intention analysis makes llms a good jailbreak defender, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.522729Z"},"links":{"cited_paper":"/paper/2401.06561","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:974c1bb92f4c94293cd370893044d9d40794179f45b847d61b2da0b3113f6b35","observation_id":"eb2b5ed2-554b-4b60-913f-271f13795738","resolution":{"observed_at":"2026-07-31T16:06:06.522729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.299","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Jailbreak open-sourced large language models via enforced decoding","venue":null,"work_id":"30652989-1154-4282-8336-bda355e9bf93","year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.459700Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:8e6120df89e35b7ffb7d23dc931c00b11d50d35270c4a1386d1e07cae8dfd61b","observation_id":"d1add4be-80e4-4697-adb3-8a270c806bb6","resolution":{"observed_at":"2026-07-31T16:11:53.860160Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-07-31T16:06:06.687929Z","title":"Why should black beans never be president?","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.687929Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:b2f3fb3fa061e1de17ad4f93f5962c8d6aa48514c01660adc4a3dc1cdda9d131","observation_id":"43f900a8-bad3-4961-a189-78e99f0bcfd0","resolution":{"observed_at":"2026-07-31T16:06:06.687929Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T16:06:06.619362Z","title":"Defending large language models against jailbreaking attacks through goal prioritization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:06.619362Z"},"links":{"citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:c4c0ca49e4cd7d79eff816aa0c7acefb3794ca70c088558d89829e19a784fb4a","observation_id":"5465ff8f-bbda-494c-8e48-909797b2a1fe","resolution":{"observed_at":"2026-07-31T16:06:06.619362Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-31T16:06:04.622310Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.622310Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:253768cf4884d7053d2ae0ed451d22f8e5e27c7f1b18a1a3a30b97f8df55891c","observation_id":"e375fec4-ae95-4a0f-9514-9a211127c156","resolution":{"observed_at":"2026-07-31T16:06:04.622310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-07T21:49:09.749932Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 0 inbound Pith citation observations for arXiv:2607.24392."}