{"as_of":"2026-08-09T06:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a0fb6fc68e4b0e2d76d382c28c281ed746d0632f03348dca9b6493ffb36a31a0","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T22:06:39.024687Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T09:29:44.257667Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"reference_index":117,"source":"pdf_text","source_observed_at":"2026-05-15T02:20:44.368219Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2407.04295"},"observation_digest":"sha256:4895c18cac57b40767f41a9fdf79fa5a3316ac5e6c05df2f05a30115b437456b","observation_id":"c586368a-150e-4f55-84a3-29d1350e9d56","resolution":{"observed_at":"2026-05-15T02:20:44.504149Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2502.01241","last_updated":"2026-04-13T12:56:05Z","snapshot_observed_at":"2026-07-06T20:30:12.663808Z","submitted_at":"2025-02-03T11:02:30Z","title":"Peering Behind the Shield: Guardrail Identification in Large Language Models","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-23T03:45:14.234545Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2502.01241"},"observation_digest":"sha256:cb719bd29c75c55033352e1ded82e5026939636de43e0a7492379a8e7adc0434","observation_id":"22481f73-652b-4e7a-abb2-ec51598fee79","resolution":{"observed_at":"2026-05-23T03:45:21.416651Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-08T22:06:39.024687Z","title":"Weak-to-strong jailbreaking on large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04643","last_updated":"2025-02-10T06:46:48Z","snapshot_observed_at":"2026-08-09T05:56:30.933327Z","submitted_at":"2025-02-07T04:07:36Z","title":"Confidence Elicitation: A New Attack Vector for Large Language Models","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-08T22:06:39.024687Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2502.04643"},"observation_digest":"sha256:c9dc0b7c7335f41a34203c9d507321752912f5df18fbe9a3345d9880e8f86fa0","observation_id":"2ed9ab0e-3aec-4809-979b-34f74362d5ad","resolution":{"observed_at":"2026-08-08T22:06:39.024687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2502.05206","last_updated":"2026-04-14T16:10:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-02T05:14:22Z","title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","version":6},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-23T04:39:04.591722Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2502.05206"},"observation_digest":"sha256:bdad00e6fb328d4b04dfeea4d5546f59a94bf008d13c01655f761fc89c29c2e0","observation_id":"e17858c5-6305-4ec4-a4cc-c9d78526c34f","resolution":{"observed_at":"2026-05-23T04:42:34.219547Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-08T10:23:06.305919Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08142","last_updated":"2025-02-12T05:48:57Z","snapshot_observed_at":"2026-08-09T00:12:26.816440Z","submitted_at":"2025-02-12T05:48:57Z","title":"Bridging the Safety Gap: A Guardrail Pipeline for Trustworthy LLM Inferences","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-08T10:23:06.305919Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2502.08142"},"observation_digest":"sha256:b2eea452a699d3f4f0ee8247f2a3502789e410117d8957f364d30b56c456baa5","observation_id":"65d59232-1a7e-4c6b-8b98-3e67cee7e60b","resolution":{"observed_at":"2026-08-08T10:23:06.305919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-07T15:41:15.103581Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14316","last_updated":"2025-05-20T13:03:15Z","snapshot_observed_at":"2026-08-08T01:19:21.267226Z","submitted_at":"2025-05-20T13:03:15Z","title":"Exploring Jailbreak Attacks on LLMs through Intent Concealment and Diversion","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T15:41:15.103581Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2505.14316"},"observation_digest":"sha256:4794964c067fbc5f966dcc90df5e4ec2a28d337ecdcef8ebe00482f01ec9ce3f","observation_id":"36ca2ae1-9574-45a8-a4bf-cc5df38bc296","resolution":{"observed_at":"2026-08-07T15:41:15.103581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-07T14:27:03.759736Z","title":"arXiv preprint arXiv:2401.17256","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.18889","last_updated":"2025-08-24T03:15:13Z","snapshot_observed_at":"2026-08-08T20:47:09.765405Z","submitted_at":"2025-05-24T22:22:43Z","title":"Security Concerns for Large Language Models: A Survey","version":5},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T14:27:03.759736Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2505.18889"},"observation_digest":"sha256:92cdc773e19e5462ea595a4a95ce31d1a7b52db4f59c7f0f0607f13e21996f3f","observation_id":"eb482b29-547d-4846-bd6b-42e9cf4b000f","resolution":{"observed_at":"2026-08-07T14:27:03.759736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-07T12:04:47.630473Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00668","last_updated":"2025-05-31T18:38:23Z","snapshot_observed_at":"2026-08-08T14:13:24.402413Z","submitted_at":"2025-05-31T18:38:23Z","title":"SafeTy Reasoning Elicitation Alignment for Multi-Turn Dialogues","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T12:04:47.630473Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2506.00668"},"observation_digest":"sha256:51465df90182c632f33eb4fb2e20974ff1d6ec273539895d04716cf9058cc3aa","observation_id":"c35c1c48-c59d-4924-9014-51bde1ac83e3","resolution":{"observed_at":"2026-08-07T12:04:47.630473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-07T05:40:20.606600Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07402","last_updated":"2025-06-09T03:52:43Z","snapshot_observed_at":"2026-08-07T10:07:41.500171Z","submitted_at":"2025-06-09T03:52:43Z","title":"Beyond Jailbreaks: Revealing Stealthier and Broader LLM Security Risks Stemming from Alignment Failures","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:40:20.606600Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2506.07402"},"observation_digest":"sha256:af24c79450522727c6969d2387b2c31ca7f9ada9bced66ff5a70ea3189d30567","observation_id":"595452c4-6357-4760-8f52-30d432d8a57b","resolution":{"observed_at":"2026-08-07T05:40:20.606600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2506.17299","last_updated":"2026-04-24T04:36:54Z","snapshot_observed_at":"2026-08-02T14:45:34.223522Z","submitted_at":"2025-06-17T20:37:29Z","title":"Toward Principled LLM Safety Testing: Solving the Jailbreak Oracle Problem","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T08:40:56.186349Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2506.17299"},"observation_digest":"sha256:9750fe2d3eccbf1220d7fffb2dbc16aeeadd2386f0e544c030825336a09f0b69","observation_id":"014a108e-00d5-492c-8dad-fb42b13f5e67","resolution":{"observed_at":"2026-05-19T08:42:12.897176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-06T23:20:57.504508Z","title":"Weak-to-strong jailbreaking on large language models.arXiv preprint arXiv:2401.172562024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.18543","last_updated":"2026-05-25T10:15:56Z","snapshot_observed_at":"2026-08-08T08:35:45.497926Z","submitted_at":"2025-06-23T11:53:31Z","title":"SoK: A Comprehensive Security Analysis of Jailbreak Resilience in GPT and DeepSeek Models","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T23:20:57.504508Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2506.18543"},"observation_digest":"sha256:458c856162d6e62f3c2850216d7e5e25b24b0de91cf1bdc4e14b0b974aa973f6","observation_id":"31368c05-277e-4bbf-b0e3-6e2c92dff6f8","resolution":{"observed_at":"2026-08-06T23:20:57.504508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-06T16:46:01.155732Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12759","last_updated":"2025-07-17T03:31:36Z","snapshot_observed_at":"2026-08-07T21:57:01.893066Z","submitted_at":"2025-07-17T03:31:36Z","title":"Logit Arithmetic Elicits Long Reasoning Capabilities Without Training","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T16:46:01.155732Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2507.12759"},"observation_digest":"sha256:cd50e6616585bf6119272c9fcff8ad7c464f28541eed2e07628942aa64c04262","observation_id":"2687ee65-1b68-46c1-91c9-905b049d5d40","resolution":{"observed_at":"2026-08-06T16:46:01.155732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-06T04:47:25.675116Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03054","last_updated":"2025-08-05T03:58:15Z","snapshot_observed_at":"2026-08-08T05:44:09.712429Z","submitted_at":"2025-08-05T03:58:15Z","title":"Beyond Surface-Level Detection: Towards Cognitive-Driven Defense Against Jailbreak Attacks via Meta-Operations Reasoning","version":1},"reference_index":196,"source":"arxiv_source","source_observed_at":"2026-08-06T04:47:25.675116Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2508.03054"},"observation_digest":"sha256:21b33225c0aa4b764e2e45a022fbf54f9a04ced05b92a8ad2fb5d1c88b645334","observation_id":"f6b8c244-c238-478b-9765-920e0081a3a9","resolution":{"observed_at":"2026-08-06T04:47:25.675116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-05T20:31:33.631918Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10404","last_updated":"2025-08-14T07:12:44Z","snapshot_observed_at":"2026-08-08T02:31:17.622343Z","submitted_at":"2025-08-14T07:12:44Z","title":"Layer-Wise Perturbations via Sparse Autoencoders for Adversarial Text Generation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T20:31:33.631918Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2508.10404"},"observation_digest":"sha256:6bc3dd713f71c2faede727bff978e3d4df781577132f39c5c329d729d338f6cc","observation_id":"cd0431ae-c85a-46f5-878f-6dcce83602e6","resolution":{"observed_at":"2026-08-05T20:31:33.631918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-04T09:25:58.581856Z","title":"Weak-to-strong jailbreaking on large language models.CoRR, abs/2401.17256, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.15476","last_updated":"2026-07-04T04:20:15Z","snapshot_observed_at":"2026-08-06T02:45:04.316908Z","submitted_at":"2025-10-17T09:38:54Z","title":"SoK: Systematizing LLM Prompt Security: Taxonomies, Datasets, and Unified Evaluation of Attacks and Defenses","version":3},"reference_index":241,"source":"pdf_text","source_observed_at":"2026-08-04T09:25:58.581856Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2510.15476"},"observation_digest":"sha256:12cd5452697bc671e90c4e7f2e306fc4e95c04bdd2ac478e766c29c05c93790a","observation_id":"4bfedeb4-c79b-4671-9818-36d308a6b432","resolution":{"observed_at":"2026-08-04T09:25:58.581856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-03T10:20:04.101125Z","title":"Andy Zou, Zifan Wang, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.10566","last_updated":"2026-05-26T12:46:14Z","snapshot_observed_at":"2026-08-08T01:21:40.454567Z","submitted_at":"2026-01-15T16:28:14Z","title":"Representation-Aware Unlearning via Activation Signatures: From Suppression to Entity-Signature Erasure","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T10:20:04.101125Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2601.10566"},"observation_digest":"sha256:22272d38bbae8dcaba5dc43bb4fb1a3c5f1c9f639c62be0ba47caae486d99b16","observation_id":"b2e2cef7-bad0-463a-8847-12988eecd8cc","resolution":{"observed_at":"2026-08-03T10:20:04.101125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2604.14602","last_updated":"2026-04-16T04:19:48Z","snapshot_observed_at":"2026-08-02T17:47:21.261876Z","submitted_at":"2026-04-16T04:19:48Z","title":"CausalDetox: Causal Head Selection and Intervention for Language Model Detoxification","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T12:04:13.987667Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2604.14602"},"observation_digest":"sha256:428f2cbff702e22d4d8f5001e6fed25d899e861390f7271c52ec5f802f7f4867","observation_id":"a4cca6e4-e5df-47f5-9d7f-d15684c99b5a","resolution":{"observed_at":"2026-05-10T12:05:21.962460Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2605.00424","last_updated":"2026-05-15T07:02:35Z","snapshot_observed_at":"2026-08-07T11:32:43.933969Z","submitted_at":"2026-05-01T05:53:05Z","title":"Skills as Verifiable Artifacts: A Trust Schema and a Biconditional Correctness Criterion for Human-in-the-Loop Agent Runtimes","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-19T18:25:12.407203Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2605.00424"},"observation_digest":"sha256:40cbe6fd69708bdbbc2eab75fe9d3caf81af2742492feb6efba380024b5af4a4","observation_id":"fe3c3664-c480-4d42-b6c2-57b112b72c48","resolution":{"observed_at":"2026-05-19T18:27:43.007360Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2606.05609","last_updated":"2026-06-04T02:31:29Z","snapshot_observed_at":"2026-08-05T14:48:48.513078Z","submitted_at":"2026-06-04T02:31:29Z","title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-28T01:16:07.252429Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2606.05609"},"observation_digest":"sha256:6ce433eed2e58f591eaa69f447326eb01efc8c59b63b1c6f651141f1c35defca","observation_id":"e0b49d19-bc3f-44ad-a27c-e72e6c28fd64","resolution":{"observed_at":"2026-07-02T13:26:59.307397Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2606.22686","last_updated":"2026-06-30T05:48:55Z","snapshot_observed_at":"2026-07-06T23:57:32.788548Z","submitted_at":"2026-06-21T22:04:48Z","title":"The Geometry of Refusal: Linear Instability in Safety-Aligned LLMs","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-26T09:52:52.603708Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2606.22686"},"observation_digest":"sha256:b9e51badcc9d0cb136cc13d06ae0d6364a62b3daf7fa5e3fe6c597019c159a3c","observation_id":"030fe290-72b6-4030-bff2-7b26a25262c6","resolution":{"observed_at":"2026-07-04T09:29:44.259446Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.17256","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-07-04T09:29:44.257667Z","title":"Weak-to-strong jailbreaking on large language models","venue":null,"work_id":"1a77a248-051b-47d2-b768-c2e9e9f7dadc","year":2025},"citing_paper":{"arxiv_id":"2606.22686","last_updated":"2026-06-30T05:48:55Z","snapshot_observed_at":"2026-07-06T23:57:32.788548Z","submitted_at":"2026-06-21T22:04:48Z","title":"The Geometry of Refusal: Linear Instability in Safety-Aligned LLMs","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-07-01T07:07:45.177242Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2606.22686"},"observation_digest":"sha256:c88e3641a387709e1a897ff9d9577b3fc6fcb863ef144b07f0c169027ba4679f","observation_id":"760ed90a-bad4-42fb-85e2-c538bd131da1","resolution":{"observed_at":"2026-07-01T08:45:35.637196Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-01T18:54:55.053927Z","title":"arXiv preprint arXiv:2401.17256 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17152","last_updated":"2026-07-19T09:18:35Z","snapshot_observed_at":"2026-08-05T14:31:41.750601Z","submitted_at":"2026-07-19T09:18:35Z","title":"How Jailbreak Attacks Inform Safety Alignment: A Defender-Centric, Shapley-Based Evaluation of Jailbreak Contributions","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-01T18:54:55.053927Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2607.17152"},"observation_digest":"sha256:652a9b0df418d4fd6aac65d40f36b310b6358b249b6a8b5b227551af3d0e28d0","observation_id":"52fadcb1-083a-450b-ba00-b19f4d4df369","resolution":{"observed_at":"2026-08-01T18:54:55.053927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17256","snapshot_observed_at":"2026-08-01T00:26:23.761661Z","title":"arXiv preprint arXiv:2401.17256 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26246","last_updated":"2026-07-28T20:31:48Z","snapshot_observed_at":"2026-08-03T10:50:56.850549Z","submitted_at":"2026-07-28T20:31:48Z","title":"Weak-to-Strong On-Policy Distillation","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-01T00:26:23.761661Z"},"links":{"cited_paper":"/paper/2401.17256","citing_paper":"/paper/2607.26246"},"observation_digest":"sha256:54ccc982c117d3ad79da6da7bfe366a19a20975e199ffaecdbc1bd81b7ff9bce","observation_id":"47529e3a-b68b-475c-a2c1-45808eaf61f4","resolution":{"observed_at":"2026-08-01T00:26:23.761661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2401.17256/citation-record","integrity":"/paper/2401.17256/integrity","json":"/paper/2401.17256/citation-record.json","paper":"/paper/2401.17256"},"outbound":[],"paper":{"arxiv_id":"2401.17256","last_updated":"2025-07-23T18:43:18Z","latest_version":5,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T17:22:41.286422Z","submitted_at":"2024-01-30T18:48:37Z","title":"Weak-to-Strong Jailbreaking on Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2401.17256."}