{"as_of":"2026-08-21T10:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2ee81ddd71cefc12a4160515233be6babc2d574e77f8daaae363aedc271dc216","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":84,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":84,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":84,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":84,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:10:30.948693Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-12T12:45:35.737949Z","title":"& Hsieh, C.-J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.16955","last_updated":"2025-02-28T21:41:21Z","snapshot_observed_at":"2026-08-19T13:30:03.079680Z","submitted_at":"2024-11-25T21:51:45Z","title":"Probing the limitations of multimodal language models for chemistry and materials research","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T12:45:35.737949Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2411.16955"},"observation_digest":"sha256:c84ee3d8f6e1b9fd5d8b6515914e8d784f47deb16ac4eb8b63bab9b46cc24c23","observation_id":"f5fbc348-805f-4b93-a464-09b387ca7d86","resolution":{"observed_at":"2026-08-12T12:45:35.737949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-11T19:25:04.775042Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.06748","last_updated":"2025-08-29T17:08:59Z","snapshot_observed_at":"2026-08-17T08:10:18.244056Z","submitted_at":"2024-12-09T18:40:44Z","title":"Refusal Tokens: A Simple Way to Calibrate Refusals in Large Language Models","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T19:25:04.775042Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2412.06748"},"observation_digest":"sha256:7d62a7cd723f33161f52b13750b153fef773449614a67744d5f1649e698372b8","observation_id":"942dcb84-786d-4c22-bd38-cb4889430a4b","resolution":{"observed_at":"2026-08-11T19:25:04.775042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-11T20:33:59.615309Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.06843","last_updated":"2024-12-11T12:35:25Z","snapshot_observed_at":"2026-08-18T16:51:27.551628Z","submitted_at":"2024-12-07T16:35:14Z","title":"Semantic Loss Guided Data Efficient Supervised Fine Tuning for Safe Responses in LLMs","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T20:33:59.615309Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2412.06843"},"observation_digest":"sha256:f9f62b1ec8addd760e19d7c920e3534c8fe1f3bc4307f1c6e26f8834a695088f","observation_id":"7490d506-06b8-4faf-97c7-129aafd999db","resolution":{"observed_at":"2026-08-11T20:33:59.615309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-10T21:24:10.025794Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04952","last_updated":"2025-01-09T03:59:10Z","snapshot_observed_at":"2026-08-18T14:04:07.234839Z","submitted_at":"2025-01-09T03:59:10Z","title":"Open Problems in Machine Unlearning for AI Safety","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T21:24:10.025794Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2501.04952"},"observation_digest":"sha256:44a0fb1c3be3745ab217b06eda7f46eeef7b65d75148e4213e276f3ee359047a","observation_id":"56584722-6f9e-41f6-9f3f-90cb256a0fd9","resolution":{"observed_at":"2026-08-10T21:24:10.025794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-10T21:18:01.315075Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05396","last_updated":"2026-07-15T02:18:29Z","snapshot_observed_at":"2026-08-14T22:15:20.283201Z","submitted_at":"2025-01-09T17:42:23Z","title":"FairCoder: Probing LLM Bias in High-Stakes Decision Making via Coding Tasks","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T21:18:01.315075Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2501.05396"},"observation_digest":"sha256:620e2733bb30aa10b1d7a4137085d1eec6690b673d786ff3ad41e018cef2a39b","observation_id":"c3d2d205-09e5-4a0b-97cd-8aebf7d70fa7","resolution":{"observed_at":"2026-08-10T21:18:01.315075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-10T19:08:01.608940Z","title":", author Chiang, W.L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.10639","last_updated":"2025-05-30T03:31:24Z","snapshot_observed_at":"2026-08-16T08:22:51.927618Z","submitted_at":"2025-01-18T02:57:12Z","title":"Latent-space adversarial training with post-aware calibration for defending large language models against jailbreak attacks","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T19:08:01.608940Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2501.10639"},"observation_digest":"sha256:3f01c689943c4c94acb001cffcaae9a747a21ebc6f1f4516600059cdec11dea7","observation_id":"c530ab1e-d492-41e7-8627-266a1c0dfe3b","resolution":{"observed_at":"2026-08-10T19:08:01.608940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-10T14:51:58.746774Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.14940","last_updated":"2026-06-28T22:17:10Z","snapshot_observed_at":"2026-08-19T21:20:11.268001Z","submitted_at":"2025-01-24T21:55:14Z","title":"CASE-Bench: Context-Aware SafEty Benchmark for Large Language Models","version":4},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-10T14:51:58.746774Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2501.14940"},"observation_digest":"sha256:7a90a464ca68f2df5e0aca27e79ee872cc77992c58e9ad1f5b36a06d319768f9","observation_id":"2c8e0039-a1d6-498f-ac5e-f3b0c5d995c2","resolution":{"observed_at":"2026-08-10T14:51:58.746774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-09T12:47:21.642807Z","title":"Or-benc h: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.02260","last_updated":"2026-06-02T17:48:48Z","snapshot_observed_at":"2026-08-18T21:16:36.251781Z","submitted_at":"2025-02-04T12:17:08Z","title":"Position: Adversarial ML for LLMs Is Not Making Any Progress","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-09T12:47:21.642807Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2502.02260"},"observation_digest":"sha256:df9b4daadde871e8deedf4dfc2f29ed8cfec2bd31d6b4a457a44c1206a8b2ba2","observation_id":"a6962c30-7769-4013-9981-9a9c3537f56b","resolution":{"observed_at":"2026-08-09T12:47:21.642807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-08T19:21:19.126985Z","title":"L., Stoica, I., & Hsieh, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06867","last_updated":"2025-02-08T04:27:33Z","snapshot_observed_at":"2026-08-15T16:52:47.335780Z","submitted_at":"2025-02-08T04:27:33Z","title":"Forbidden Science: Dual-Use AI Challenge Benchmark and Scientific Refusal Tests","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T19:21:19.126985Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2502.06867"},"observation_digest":"sha256:a50b575de7d879e6689865cf545e5a7eaf390b63fd868b0a95818cd5d74fd69f","observation_id":"e67b5b15-8613-41b6-8f52-7b5cf9eaefc6","resolution":{"observed_at":"2026-08-08T19:21:19.126985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-08T12:25:30.547187Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07557","last_updated":"2025-02-11T13:50:50Z","snapshot_observed_at":"2026-08-19T00:16:08.532848Z","submitted_at":"2025-02-11T13:50:50Z","title":"JBShield: Defending Large Language Models from Jailbreak Attacks through Activated Concept Analysis and Manipulation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T12:25:30.547187Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2502.07557"},"observation_digest":"sha256:64f2bc9dcf2bec2aea3aa0eb8ffdc59f09dc5de892908995ea7b978acd794785","observation_id":"1700476e-fa02-4fb5-b33e-7ea86dd76dd5","resolution":{"observed_at":"2026-08-08T12:25:30.547187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T23:08:02.240706Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09674","last_updated":"2025-05-27T08:40:42Z","snapshot_observed_at":"2026-08-17T02:07:51.696583Z","submitted_at":"2025-02-13T06:39:22Z","title":"The Hidden Dimensions of LLM Alignment: A Multi-Dimensional Analysis of Orthogonal Safety Directions","version":4},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T23:08:02.240706Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2502.09674"},"observation_digest":"sha256:67edf5887e4d191798985d213622652b2a2617554188dda2bb9123fd6865e4bb","observation_id":"9b20750b-9e2c-47e2-ad85-3430e0d371e9","resolution":{"observed_at":"2026-08-07T23:08:02.240706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2503.02574","last_updated":"2026-05-18T17:54:05Z","snapshot_observed_at":"2026-08-18T01:04:38.795907Z","submitted_at":"2025-03-04T12:55:07Z","title":"LLM-Safety Evaluations Lack Robustness","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-23T01:26:45.402983Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2503.02574"},"observation_digest":"sha256:1e20d207225d43d5af18c8c608a029e17396f2accd50912779e2af3147ba1c9a","observation_id":"b55a4152-bee3-447d-aa84-fe646c1e8a6b","resolution":{"observed_at":"2026-05-23T01:27:21.424406Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-16T12:10:30.948693Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.13562","last_updated":"2025-04-18T09:02:12Z","snapshot_observed_at":"2026-08-20T14:24:36.279935Z","submitted_at":"2025-04-18T09:02:12Z","title":"DETAM: Defending LLMs Against Jailbreak Attacks via Targeted Attention Modification","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-16T12:10:30.948693Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2504.13562"},"observation_digest":"sha256:afe3a813dc2e04f0dfd12acbbcbdfe5a082a640656f0d01fb3f12b96c6267e4c","observation_id":"a90b09ef-8ac4-4e71-876e-b7d460e28410","resolution":{"observed_at":"2026-08-16T12:10:30.948693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-16T11:39:10.716090Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14985","last_updated":"2025-04-23T16:52:54Z","snapshot_observed_at":"2026-08-19T05:45:48.314083Z","submitted_at":"2025-04-21T09:26:05Z","title":"aiXamine: Simplified LLM Safety and Security","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T11:39:10.716090Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2504.14985"},"observation_digest":"sha256:504b24e96c57adb132c07d911ecc3d09defd867106aa4ba2441c9e2ab6e9e357","observation_id":"a6fd153e-26cb-405d-a1e7-45419fbf37a7","resolution":{"observed_at":"2026-08-16T11:39:10.716090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-15T22:10:09.270342Z","title":"Justin Cui, Wei-Lin Chiang, Ion Stoica, and Cho-Jui Hsieh","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.08054","last_updated":"2025-07-15T10:03:15Z","snapshot_observed_at":"2026-08-19T20:05:11.167007Z","submitted_at":"2025-05-12T20:45:25Z","title":"FalseReject: A Resource for Improving Contextual Safety and Mitigating Over-Refusals in LLMs via Structured Reasoning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T22:10:09.270342Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2505.08054"},"observation_digest":"sha256:1c8c4bb3aaf239ab44b6410c8b4c2782b31492a070687b98ffbb97797de36ddd","observation_id":"0fa683eb-bdc3-41b1-8e9d-19f464270ddd","resolution":{"observed_at":"2026-08-15T22:10:09.270342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T15:12:19.066466Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17131","last_updated":"2025-05-22T01:59:54Z","snapshot_observed_at":"2026-08-16T01:50:50.437369Z","submitted_at":"2025-05-22T01:59:54Z","title":"Relative Bias: A Comparative Framework for Quantifying Bias in LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:12:19.066466Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2505.17131"},"observation_digest":"sha256:be399b39073a205563539e97da55f5edd738ba081cd5eafa63771d75695e1363","observation_id":"0758b737-4732-4884-9924-d4b18d25fad9","resolution":{"observed_at":"2026-08-07T15:12:19.066466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T14:13:52.566860Z","title":"Or-bench: An over-refusal benchmark for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19690","last_updated":"2025-05-26T08:49:19Z","snapshot_observed_at":"2026-08-17T01:37:19.231497Z","submitted_at":"2025-05-26T08:49:19Z","title":"Beyond Safe Answers: A Benchmark for Evaluating True Risk Awareness in Large Reasoning Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:13:52.566860Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2505.19690"},"observation_digest":"sha256:12b20cfa3adc446d1f572ddcca1823319c6cce042d5c4aa78a3b6964d2043fca","observation_id":"3ada40e2-6b9b-427a-82a7-f2956c25f27c","resolution":{"observed_at":"2026-08-07T14:13:52.566860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T14:13:10.702663Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20362","last_updated":"2025-05-26T09:01:46Z","snapshot_observed_at":"2026-08-19T14:44:52.477941Z","submitted_at":"2025-05-26T09:01:46Z","title":"VSCBench: Bridging the Gap in Vision-Language Model Safety Calibration","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T14:13:10.702663Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2505.20362"},"observation_digest":"sha256:0d6942e4fe501d40a02f39d1805531a3fbdd169bf889ba3704d130ac9329668c","observation_id":"85030306-0453-4f53-ba6c-4f12428412b5","resolution":{"observed_at":"2026-08-07T14:13:10.702663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T13:52:09.068357Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20925","last_updated":"2025-05-27T09:15:03Z","snapshot_observed_at":"2026-08-17T23:21:15.835136Z","submitted_at":"2025-05-27T09:15:03Z","title":"Multi-objective Large Language Model Alignment with Hierarchical Experts","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T13:52:09.068357Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2505.20925"},"observation_digest":"sha256:2e7d59c1752a2ebb9568fcb2ede4bbd053855b543f29712c0751351cab2becbe","observation_id":"2af4f3a5-2938-40c3-be36-2ef53c711183","resolution":{"observed_at":"2026-08-07T13:52:09.068357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2506.01770","last_updated":"2026-04-18T05:40:12Z","snapshot_observed_at":"2026-08-15T02:56:03.854072Z","submitted_at":"2025-06-02T15:17:38Z","title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-19T11:34:09.428653Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2506.01770"},"observation_digest":"sha256:e143faf342603bc0123a174a0e91c8099e11bcbccd3df10a45980f2506cc3a50","observation_id":"bd4cb0c8-2ab2-4de2-ac79-4773c7df3b74","resolution":{"observed_at":"2026-05-19T11:37:15.914531Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T11:29:01.068351Z","title":"In Forty-first International Conference on Machine Learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02460","last_updated":"2025-06-03T05:23:09Z","snapshot_observed_at":"2026-08-20T22:01:45.082797Z","submitted_at":"2025-06-03T05:23:09Z","title":"MidPO: Dual Preference Optimization for Safety and Helpfulness in Large Language Models via a Mixture of Experts Framework","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T11:29:01.068351Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2506.02460"},"observation_digest":"sha256:19050d3879a8a85b4fbf72abc6ab9feb5c335d05db26820b7830fbc235600e06","observation_id":"f366d40f-8412-423a-8efd-6d89fba3a5ef","resolution":{"observed_at":"2026-08-07T11:29:01.068351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T10:28:54.986950Z","title":"Cui, W.-L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06391","last_updated":"2025-06-05T16:53:29Z","snapshot_observed_at":"2026-08-14T22:15:17.582406Z","submitted_at":"2025-06-05T16:53:29Z","title":"From Rogue to Safe AI: The Role of Explicit Refusals in Aligning LLMs with International Humanitarian Law","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:54.986950Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2506.06391"},"observation_digest":"sha256:56c88de1222faaad0233738d780c0692b153b65e27709c6390d44a00d6787c41","observation_id":"14de37fd-3c28-4821-adbe-efc2d4d0a818","resolution":{"observed_at":"2026-08-07T10:28:54.986950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-07T05:18:59.574137Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08399","last_updated":"2025-06-11T06:57:37Z","snapshot_observed_at":"2026-08-16T22:00:46.980531Z","submitted_at":"2025-06-10T03:13:50Z","title":"SafeCoT: Improving VLM Safety with Minimal Reasoning","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T05:18:59.574137Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2506.08399"},"observation_digest":"sha256:1201952a6f022e4cbcc39b7169101bdea0db6d8b382b65e3bd48ecf61896e994","observation_id":"d3887f60-c7e8-49ca-b717-bf18f935a22d","resolution":{"observed_at":"2026-08-07T05:18:59.574137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-06T19:57:43.629886Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04250","last_updated":"2025-07-06T05:47:04Z","snapshot_observed_at":"2026-08-13T03:26:28.377277Z","submitted_at":"2025-07-06T05:47:04Z","title":"Just Enough Shifts: Mitigating Over-Refusal in Aligned Language Models with Targeted Representation Fine-Tuning","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T19:57:43.629886Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2507.04250"},"observation_digest":"sha256:d39e52c5e195f04ba17f6f82d5f36744ed4c5ff6dbfc46639120ac5185ab8655","observation_id":"245fc8a6-5db8-4ccb-8ce6-e13192ca047c","resolution":{"observed_at":"2026-08-06T19:57:43.629886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-06T17:03:15.633680Z","title":"Or-bench: An over-refusal benchmark for large language models.arXiv preprint arXiv:2405.20947,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11878","last_updated":"2026-07-06T01:46:44Z","snapshot_observed_at":"2026-08-13T20:49:32.387386Z","submitted_at":"2025-07-16T03:48:03Z","title":"LLMs Encode Harmfulness and Refusal Separately","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:03:15.633680Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2507.11878"},"observation_digest":"sha256:e7490a96a9e194973a6f76358a9b364873f87268e7505cd57dbdb732c4ea0f3c","observation_id":"cae4b572-6abc-48c8-bd49-80ef44374df2","resolution":{"observed_at":"2026-08-06T17:03:15.633680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-06T04:47:25.517815Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03054","last_updated":"2025-08-05T03:58:15Z","snapshot_observed_at":"2026-08-15T06:04:21.097898Z","submitted_at":"2025-08-05T03:58:15Z","title":"Beyond Surface-Level Detection: Towards Cognitive-Driven Defense Against Jailbreak Attacks via Meta-Operations Reasoning","version":1},"reference_index":149,"source":"arxiv_source","source_observed_at":"2026-08-06T04:47:25.517815Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2508.03054"},"observation_digest":"sha256:fc7fc51c70b3fdefa8611cf2c7370e025520447d3bc6bec5f04adb48e80ebc2f","observation_id":"553347cf-8323-4ce0-8d54-573938d47ef4","resolution":{"observed_at":"2026-08-06T04:47:25.517815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2508.11222","last_updated":"2025-12-05T05:36:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-15T05:03:26Z","title":"ORFuzz: Fuzzing the \"Other Side\" of LLM Safety -- Testing Over-Refusal","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T23:27:52.438709Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2508.11222"},"observation_digest":"sha256:c20b8a8982062a71a56339ae6ab84fee84049912d97aae1017514a2a42a990e8","observation_id":"2a28ee14-ff1e-4212-a162-936697886bee","resolution":{"observed_at":"2026-05-18T23:31:54.602048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-04T21:22:14.290904Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08075","last_updated":"2025-09-09T18:30:01Z","snapshot_observed_at":"2026-08-12T18:58:31.147712Z","submitted_at":"2025-09-09T18:30:01Z","title":"No for Some, Yes for Others: Persona Prompts and Other Sources of False Refusal in Language Models","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-04T21:22:14.290904Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2509.08075"},"observation_digest":"sha256:b98b9edfd4b384baea5d55011adbc1eb2d5d4456dfb6ac5f6be52e4807891334","observation_id":"6dc443b8-3a12-41f1-addc-a74c324002bf","resolution":{"observed_at":"2026-08-04T21:22:14.290904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-03T12:42:38.838889Z","title":"OR-Bench: An over-refusal benchmark for large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.02023","last_updated":"2026-07-14T21:47:31Z","snapshot_observed_at":"2026-08-17T09:16:55.559134Z","submitted_at":"2026-01-05T11:30:56Z","title":"Not All Needles Are Found: How Fact Distribution and Don't Make It Up Prompts Shape Retrieval, Reasoning, and Hallucination in Long-Context LLMs","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-03T12:42:38.838889Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2601.02023"},"observation_digest":"sha256:7b7395db1c454520957979c1699669ba3a14f310bc1f3dfd893d6e93a3d84270","observation_id":"93cf701d-1912-4476-9664-f2189ee9d5b2","resolution":{"observed_at":"2026-08-03T12:42:38.838889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-03T11:35:39.627069Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.05751","last_updated":"2026-06-05T07:06:47Z","snapshot_observed_at":"2026-08-04T17:00:12.286204Z","submitted_at":"2026-01-09T12:07:38Z","title":"Analysing Differences in Persuasive Language in LLM-Generated Text: Uncovering Stereotypical Gender Patterns","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-03T11:35:39.627069Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2601.05751"},"observation_digest":"sha256:3c24ff097047d3418151900cacc066270743ac70f843d7afe1f0050b24e9d891","observation_id":"871570c7-db20-4069-af6f-8f8071b41550","resolution":{"observed_at":"2026-08-03T11:35:39.627069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2602.02280","last_updated":"2026-05-12T03:47:11Z","snapshot_observed_at":"2026-08-11T07:06:38.718933Z","submitted_at":"2026-02-02T16:20:51Z","title":"RACC: Representation-Aware Coverage Criteria for LLM Safety Testing","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T08:12:55.296932Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2602.02280"},"observation_digest":"sha256:3063a85f11d67bd6c7a2f6690e9c55597092bb766c8f60e4e34e7109a54f287d","observation_id":"7750d229-a914-4ecf-96d5-e216c4dfecc0","resolution":{"observed_at":"2026-05-16T08:17:36.646432Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2602.08813","last_updated":"2026-05-12T17:22:35Z","snapshot_observed_at":"2026-08-13T02:15:10.817007Z","submitted_at":"2026-02-09T15:50:05Z","title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T05:33:42.965249Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2602.08813"},"observation_digest":"sha256:38cfeeead80f91614f63399b211c11da48c2d772b6851f8deb85872cbe1853fc","observation_id":"1c22e9c0-1e31-49c9-85d9-05b0b27cb9f3","resolution":{"observed_at":"2026-05-16T05:37:24.313558Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-02T21:09:49.563604Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.21556","last_updated":"2026-07-08T07:53:59Z","snapshot_observed_at":"2026-08-19T22:21:43.426289Z","submitted_at":"2026-02-25T04:23:50Z","title":"Power and Limitations of Aggregation in Compound AI Systems","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-02T21:09:49.563604Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2602.21556"},"observation_digest":"sha256:7aad65db9a6b85bb09c81966162ec83370b09827fcc808c0198ceb86d5d36744","observation_id":"cf3358a2-13d2-4260-a988-dfd0c2063134","resolution":{"observed_at":"2026-08-02T21:09:49.563604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2604.01473","last_updated":"2026-05-28T05:44:14Z","snapshot_observed_at":"2026-08-15T03:04:08.722858Z","submitted_at":"2026-04-01T23:29:12Z","title":"SelfGrader: LLM Jailbreak Detection via Anchored Token-Level Logits","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T21:43:46.728512Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2604.01473"},"observation_digest":"sha256:f3819e0b1d92a0d6033f0f14da135dc5748b9de6deb944a5668a28f04125887f","observation_id":"02b49cf3-1707-4c19-a26b-164617b8a9ef","resolution":{"observed_at":"2026-05-13T21:48:19.437868Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2604.06233","last_updated":"2026-04-03T13:53:23Z","snapshot_observed_at":"2026-08-17T19:17:24.958879Z","submitted_at":"2026-04-03T13:53:23Z","title":"Blind Refusal: Language Models Refuse to Help Users Evade Unjust, Absurd, and Illegitimate Rules","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-13T19:24:54.381722Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2604.06233"},"observation_digest":"sha256:8593658aee433825ecf19794f354c2a568892cdd7a9ba404e674ed1f6b23c652","observation_id":"c62b8c9d-7145-4eb7-a75e-aa14cfe5e50e","resolution":{"observed_at":"2026-05-13T19:28:09.938746Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2604.07709","last_updated":"2026-06-03T21:15:24Z","snapshot_observed_at":"2026-08-17T00:44:55.340054Z","submitted_at":"2026-04-09T01:54:33Z","title":"IatroBench: Pre-Registered Evidence of Iatrogenic Harm from AI Safety Measures","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-10T18:25:53.037936Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2604.07709"},"observation_digest":"sha256:0f8ad9f99c55edff125625fbde53cc2c61d8c185505ab996304378d152ee0447","observation_id":"76979c98-1866-4261-ad43-e1aa012fc0f1","resolution":{"observed_at":"2026-05-11T00:35:52.516530Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-07-13T00:19:33.861692Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.07709","last_updated":"2026-06-03T21:15:24Z","snapshot_observed_at":"2026-08-17T00:44:55.340054Z","submitted_at":"2026-04-09T01:54:33Z","title":"IatroBench: Pre-Registered Evidence of Iatrogenic Harm from AI Safety Measures","version":4},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-07-13T00:19:33.861692Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2604.07709"},"observation_digest":"sha256:d467857a89c140c690f229d35b9afad4d61e30209593e61cdf0b41ddda93ac32","observation_id":"fac9a0e5-6f66-49ac-8b5b-ccb6562af93c","resolution":{"observed_at":"2026-07-13T00:19:33.861692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2604.19049","last_updated":"2026-04-21T03:55:35Z","snapshot_observed_at":"2026-08-21T08:17:55.760209Z","submitted_at":"2026-04-21T03:55:35Z","title":"Refute-or-Promote: An Adversarial Stage-Gated Multi-Agent Review Methodology for High-Precision LLM-Assisted Defect Discovery","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T03:11:26.513954Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2604.19049"},"observation_digest":"sha256:cbb596de74e24ab4564b8e306a45e5f0244d6e5fd58c5d44f64da9c6d176e524","observation_id":"bb565363-ccb2-4e0f-be8f-cf4d6bcbb223","resolution":{"observed_at":"2026-05-10T03:14:08.024390Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2604.25110","last_updated":"2026-05-05T18:12:18Z","snapshot_observed_at":"2026-08-06T08:02:49.193188Z","submitted_at":"2026-04-28T01:32:46Z","title":"Knowledge Distillation Must Account for What It Loses","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-07T16:58:41.449172Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2604.25110"},"observation_digest":"sha256:d597a215f859b8930a3b39fd2d2f157e04f3b3edd8b5e37b1435f686e9b964d1","observation_id":"cc58d492-46c4-4f46-9699-b298b3cc5160","resolution":{"observed_at":"2026-05-11T23:26:18.660235Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2604.25110","last_updated":"2026-05-05T18:12:18Z","snapshot_observed_at":"2026-08-06T08:02:49.193188Z","submitted_at":"2026-04-28T01:32:46Z","title":"Knowledge Distillation Must Account for What It Loses","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-08T03:31:56.787201Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2604.25110"},"observation_digest":"sha256:8e16415392b75109b55f044038616ec5f340186d78fca313e96dc73b65ba8f3f","observation_id":"a0c6fad7-ca75-4fce-9784-7c3450e92ee2","resolution":{"observed_at":"2026-05-11T22:01:13.526306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.01899","last_updated":"2026-05-03T14:28:08Z","snapshot_observed_at":"2026-08-17T00:31:30.183586Z","submitted_at":"2026-05-03T14:28:08Z","title":"Disentangling Intent from Role: Adversarial Self-Play for Persona-Invariant Safety Alignment","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-09T17:24:54.796037Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.01899"},"observation_digest":"sha256:99fad877ae7aa7ea5cccda2ce8f4dd0c10646d4146024606b965b6d3daaf150c","observation_id":"5f9105b3-2973-49d7-abaa-e88601ad6349","resolution":{"observed_at":"2026-05-11T16:21:07.419924Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.08496","last_updated":"2026-05-08T21:21:59Z","snapshot_observed_at":"2026-08-20T16:03:51.183472Z","submitted_at":"2026-05-08T21:21:59Z","title":"Latent Personality Alignment: Improving Harmlessness Without Mentioning Harms","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T01:46:49.586630Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.08496"},"observation_digest":"sha256:95b8e924a12a34a53c8295c335830127a31309f4ea5ee9c5eaa40d47da035b18","observation_id":"7a09cf8e-8d6d-4c43-ac3a-5df817ba9e1f","resolution":{"observed_at":"2026-05-12T07:51:43.562457Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.09278","last_updated":"2026-05-10T03:04:12Z","snapshot_observed_at":"2026-08-14T15:06:21.586001Z","submitted_at":"2026-05-10T03:04:12Z","title":"EquiMem: Calibrating Shared Memory in Multi-Agent Debate via Game-Theoretic Equilibrium","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T04:47:29.903343Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.09278"},"observation_digest":"sha256:85812b296c86e174cb8728aa49b56eef1d38a43c11798c1fb8b92919e6ddd189","observation_id":"89801801-ac87-427f-9977-2cfc2f76e90a","resolution":{"observed_at":"2026-05-12T05:56:26.309530Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.16282","last_updated":"2026-04-11T04:25:19Z","snapshot_observed_at":"2026-08-12T21:07:09.004725Z","submitted_at":"2026-04-11T04:25:19Z","title":"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-21T01:42:55.693115Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.16282"},"observation_digest":"sha256:93d7dfa7ee2541e9c52384621107cdda298ed2f09b2672efe2ffc14f4ee7929b","observation_id":"2cbda156-ec48-46de-92e9-fed450dadf5a","resolution":{"observed_at":"2026-05-21T01:43:56.924885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:2b3f69a25d47de488493e4dab6a4841936f32a09859ba9b4f02744d75e2825e9","observation_id":"757b7704-4f27-4c1d-9dd3-cfca173d462b","resolution":{"observed_at":"2026-05-22T01:20:51.889010Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.24154","last_updated":"2026-05-22T19:22:17Z","snapshot_observed_at":"2026-08-14T08:11:37.614977Z","submitted_at":"2026-05-22T19:22:17Z","title":"Palette: A Modular, Controllable, and Efficient Framework for On-demand Authorized Safety Alignment Relaxation in LLMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-30T16:03:12.728352Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.24154"},"observation_digest":"sha256:ae83f57c6754baeab874e3680a3cade0456a792ed8b87090d98bfcd761199d7d","observation_id":"5af05fa0-c542-49d7-9279-4c1efac0af7e","resolution":{"observed_at":"2026-06-30T16:04:52.628366Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.24552","last_updated":"2026-05-23T12:39:25Z","snapshot_observed_at":"2026-08-14T00:49:41.018461Z","submitted_at":"2026-05-23T12:39:25Z","title":"Ellipsoid Control: A White-list Jailbreak Defense via Benign Latent Modeling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-30T13:10:44.728498Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.24552"},"observation_digest":"sha256:5a3c4c0d97b8d9b60d6514313ce52168cb285974f3048fe6ca4cf9944befe3a7","observation_id":"5c988fe2-ee8a-4724-ac32-bb8ab070abe3","resolution":{"observed_at":"2026-06-30T13:14:40.779901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.28647","last_updated":"2026-05-27T15:52:07Z","snapshot_observed_at":"2026-08-15T04:16:11.072271Z","submitted_at":"2026-05-27T15:52:07Z","title":"The Ethics of LLM Sandbox and Persona Dynamics","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T12:42:48.734781Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.28647"},"observation_digest":"sha256:5aa111769f43f1af156941b3a001622912bd65921a7b725d71443644ea5a0dee","observation_id":"416ab8e3-2843-41ef-b8fb-b57381d80de1","resolution":{"observed_at":"2026-06-29T12:43:25.126621Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.29659","last_updated":"2026-05-28T09:21:42Z","snapshot_observed_at":"2026-08-12T14:52:35.122614Z","submitted_at":"2026-05-28T09:21:42Z","title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T09:11:58.843585Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.29659"},"observation_digest":"sha256:8f9e62fd314d19f07fbb0e363a9cda8bc3d56fa25ac5849725cef47c8df05fae","observation_id":"bb8d5d0a-b010-4562-95f8-aef7057d4e39","resolution":{"observed_at":"2026-06-29T09:13:15.991497Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.30693","last_updated":"2026-05-29T00:36:48Z","snapshot_observed_at":"2026-08-14T21:42:23.598641Z","submitted_at":"2026-05-29T00:36:48Z","title":"Triaging Threats to Specialized Guardrails","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-28T22:27:44.703466Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.30693"},"observation_digest":"sha256:b165a38732c9636a28e88ab504724b74c4f65ca60975541b4003798b8af50d60","observation_id":"bb7074f5-398f-4098-bd1a-22c6caf3c5ac","resolution":{"observed_at":"2026-06-28T22:32:44.346641Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.00600","last_updated":"2026-05-30T07:57:56Z","snapshot_observed_at":"2026-08-08T17:35:26.016727Z","submitted_at":"2026-05-30T07:57:56Z","title":"Understanding the Self-Reflection Mechanisms of LLMs through Biased Attitude Associations","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-28T18:17:18.706689Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.00600"},"observation_digest":"sha256:340ccd2c9a8f80307c3a40be1b23f89fe674c6ea875caf2ce0656c419541c841","observation_id":"48e132d9-c9a1-4161-b6c8-3f39f5c8e78b","resolution":{"observed_at":"2026-07-01T20:36:12.167986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.00686","last_updated":"2026-05-30T11:49:42Z","snapshot_observed_at":"2026-08-04T22:47:09.122420Z","submitted_at":"2026-05-30T11:49:42Z","title":"Dialectics of Alignment: Harnessing Unsafe Knowledge for Dynamic Safety Routing","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T19:23:10.275977Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.00686"},"observation_digest":"sha256:a20b46a884152eb088061ee5a72a9f326a72293c1dbb3e9e63995a5c151a6479","observation_id":"ab1d4848-967d-44de-8620-232ccbf4e0c7","resolution":{"observed_at":"2026-06-28T19:32:34.575384Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.02965","last_updated":"2026-08-03T20:59:18Z","snapshot_observed_at":"2026-08-07T23:09:28.856830Z","submitted_at":"2026-06-01T23:52:56Z","title":"Designing for Doubt: The Case for Informed Abstention in Autonomous Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T14:03:02.703499Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.02965"},"observation_digest":"sha256:22002ec870c034ed9a8ef7f0978956d49f69678a45842d07cfab2e7e41e6dbc3","observation_id":"f8aebd10-570b-44a8-9058-8b3904623bdf","resolution":{"observed_at":"2026-07-01T23:46:23.700019Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.03330","last_updated":"2026-06-02T08:39:50Z","snapshot_observed_at":"2026-08-02T20:44:01.368085Z","submitted_at":"2026-06-02T08:39:50Z","title":"FLIPS: Instance-Fingerprinting for LLMs via Pseudo-random Sequences","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-06-28T11:24:01.547119Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.03330"},"observation_digest":"sha256:6e926dde81136db5959bce7fb3d27bd780a4a953651d70863b3157a255d0c775","observation_id":"f98cd193-7c9e-47f8-aa79-a8edcb84d1bf","resolution":{"observed_at":"2026-07-02T01:56:27.942088Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.05523","last_updated":"2026-06-04T00:06:13Z","snapshot_observed_at":"2026-08-08T00:57:41.523461Z","submitted_at":"2026-06-04T00:06:13Z","title":"CHASE: Adversarial Red-Blue Teaming for Improving LLM Safety using Reinforcement Learning","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-06-28T02:34:26.334078Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.05523"},"observation_digest":"sha256:50952e08d2666200e935b5b7c958c0b9c9d525b5de9ad2eb331839963725f47e","observation_id":"ad9524b8-ea57-4769-8479-65d9d013ab76","resolution":{"observed_at":"2026-07-02T12:06:55.639498Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.08381","last_updated":"2026-06-07T00:20:55Z","snapshot_observed_at":"2026-08-12T22:28:39.757023Z","submitted_at":"2026-06-07T00:20:55Z","title":"Auditing Proprietary Alignment in Large Language Models: A Comparative Framework Without a Ground-Truth Standard","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T19:04:07.735560Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.08381"},"observation_digest":"sha256:627b063733f52946df9c4899bdf74ac5b53a15e5582e6b6351f026d48f9e9eca","observation_id":"4246cc48-4cb6-4ab8-93a2-1fba21e54237","resolution":{"observed_at":"2026-07-02T22:17:26.020107Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.09697","last_updated":"2026-06-08T16:19:18Z","snapshot_observed_at":"2026-07-31T09:20:59.080190Z","submitted_at":"2026-06-08T16:19:18Z","title":"PsychoSafe: Eliciting Psychologically-Informed Refusals in Large Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T16:31:33.520989Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.09697"},"observation_digest":"sha256:676d4d4f419d93c476a87bdc29be20977bf93536ecafe35177ef794b1f9db3b9","observation_id":"d3931b56-9828-4521-9892-854c1e7ed77b","resolution":{"observed_at":"2026-07-03T01:27:31.228542Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.09843","last_updated":"2026-07-07T00:32:51Z","snapshot_observed_at":"2026-08-03T03:55:17.858474Z","submitted_at":"2026-04-24T04:42:09Z","title":"An LLM-Native Psychometric Instrument Reveals a Self-Report--Behavior Gap Across 25 Models","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-07-04T19:46:24.142611Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.09843"},"observation_digest":"sha256:e4f32760e6f31a0fabdd78fe0c1d284f08eb75cc4853933e1c39cac1b5cc072b","observation_id":"128ae228-7814-421e-95c1-653d2b133968","resolution":{"observed_at":"2026-07-04T19:50:09.966245Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.11316","last_updated":"2026-06-09T18:01:19Z","snapshot_observed_at":"2026-08-01T20:26:14.151904Z","submitted_at":"2026-06-09T18:01:19Z","title":"Sch\\\"utzen: Evaluating LLM Safety in Bulgarian and German Contexts","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-27T13:32:18.368158Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.11316"},"observation_digest":"sha256:4cfd1ffa4e6bc16cfd80b70280f93c2c680aa226df39ac75c63d7f8a2851b04b","observation_id":"02c3ae61-49c3-4c53-967b-73e436e652e1","resolution":{"observed_at":"2026-07-03T04:57:38.390307Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.12429","last_updated":"2026-05-14T23:12:14Z","snapshot_observed_at":"2026-08-14T06:23:01.048019Z","submitted_at":"2026-05-14T23:12:14Z","title":"Muse Spark Safety & Preparedness Report","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-06-30T19:49:14.463992Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.12429"},"observation_digest":"sha256:7b0643975436bf497c2c2a22cbd3efcb4cafdd43e845fe16b52ef22d82abeaee","observation_id":"829039a5-9561-4f0f-8276-604a52c9b12d","resolution":{"observed_at":"2026-06-30T19:55:01.558009Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-07-12T13:51:23.302296Z","title":"OR-Bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.16349","last_updated":"2026-07-06T07:55:59Z","snapshot_observed_at":"2026-08-16T14:52:55.815325Z","submitted_at":"2026-06-15T07:50:00Z","title":"From Refusal Geometry to Safety Geometry: Harmfulness--Refusal Coupling under Dynamic Adversarial Fine-Tuning","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-12T13:51:23.302296Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.16349"},"observation_digest":"sha256:f5776052eda1e97760b01c19d140acf3b224fb7823c2312fec2e23035d3b5564","observation_id":"8e0ad023-5950-4a2d-9d13-8aaed034aadb","resolution":{"observed_at":"2026-07-12T13:51:23.302296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.21296","last_updated":"2026-06-19T10:19:54Z","snapshot_observed_at":"2026-08-02T18:19:53.372016Z","submitted_at":"2026-06-19T10:19:54Z","title":"Discriminatory Compliance: How LLMs Answer Queries from Protected Groups","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-06-26T12:58:09.227648Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.21296"},"observation_digest":"sha256:7bb1e78479a8f6f9cc0b44c6e76fdde2d59e2d588cf842bfbb89855957f508cd","observation_id":"9af1926e-30cb-4dd5-ac00-53824bc8a398","resolution":{"observed_at":"2026-06-26T12:59:29.372405Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.26396","last_updated":"2026-06-24T21:26:43Z","snapshot_observed_at":"2026-08-02T16:34:16.438122Z","submitted_at":"2026-06-24T21:26:43Z","title":"At the Edge of Understanding: Sparse Autoencoders Trace The Limits of Transformer Generalization","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-26T01:27:39.812228Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.26396"},"observation_digest":"sha256:e9e47835256d1019659114bd2fbd357f0fdee62da5852adb59ece79ec1852ddb","observation_id":"5b5acc47-0346-407d-8bf6-be5966ab85d1","resolution":{"observed_at":"2026-07-04T15:49:56.284738Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2606.31748","last_updated":"2026-06-30T14:38:49Z","snapshot_observed_at":"2026-07-07T00:05:27.130060Z","submitted_at":"2026-06-30T14:38:49Z","title":"Addressing Over-Refusal in LLMs with Competing Rewards","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-07-01T06:59:12.695984Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2606.31748"},"observation_digest":"sha256:aa3d1be79f363dc5de06a3511d13a70b76ced348504e8141e992392fd3d25197","observation_id":"92957cf9-1ae4-4cc4-b834-f68eedbeaf05","resolution":{"observed_at":"2026-07-01T08:55:35.559051Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2607.02047","last_updated":"2026-07-02T11:14:52Z","snapshot_observed_at":"2026-08-18T14:10:49.713931Z","submitted_at":"2026-07-02T11:14:52Z","title":"OpenSafeIntent: Evaluating Intent-Calibrated Safe Completion Across Dual-Use Prompt Sets","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-07-03T14:44:57.205766Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.02047"},"observation_digest":"sha256:b4cf31905cd41ac6997658e1c28c1782a04695143c62b1586a73559642a1aede","observation_id":"2dff7a81-6bd7-42df-a1c0-b554c3131951","resolution":{"observed_at":"2026-07-03T14:48:32.499370Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2607.05355","last_updated":"2026-07-06T17:33:36Z","snapshot_observed_at":"2026-07-09T23:18:25.416188Z","submitted_at":"2026-07-06T17:33:36Z","title":"Faithfulness to Refusal: A Causal Audit of Neuron Selectors","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-07T15:35:54.665268Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.05355"},"observation_digest":"sha256:640db42a1881bc6ad2cb9d889fb959770724f902b19782bb19bb86e2fda967c2","observation_id":"b951f89c-121b-4456-b48b-850bf11e9d12","resolution":{"observed_at":"2026-07-07T15:43:53.889765Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-07-11T16:42:49.284801Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05462","last_updated":"2026-07-21T17:05:39Z","snapshot_observed_at":"2026-08-18T12:55:08.383819Z","submitted_at":"2026-07-06T01:46:07Z","title":"BioSecBench-Refusal: A paired metric for performance and alignment in agentic biosecurity risk assessment","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-11T16:42:49.284801Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.05462"},"observation_digest":"sha256:a8b3f0f54cd241b7b8dc05b76508b3265e82d2e3dd42a84cb34cc50d6b5f7107","observation_id":"b1e0bd40-6d38-48f1-b9a5-2a90a79377b5","resolution":{"observed_at":"2026-07-11T16:42:49.284801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-02T08:40:01.376681Z","title":"Or-bench: An over-refusal benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05462","last_updated":"2026-07-21T17:05:39Z","snapshot_observed_at":"2026-08-18T12:55:08.383819Z","submitted_at":"2026-07-06T01:46:07Z","title":"BioSecBench-Refusal: A paired metric for performance and alignment in agentic biosecurity risk assessment","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T08:40:01.376681Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.05462"},"observation_digest":"sha256:ff1dfb6b09e67f676d4ad1907ed70658b4b4c2a4b1a1a6ce5cdbb8712921af09","observation_id":"22a13f1a-389d-48ae-8847-14d41dbd3c08","resolution":{"observed_at":"2026-08-02T08:40:01.376681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2607.05842","last_updated":"2026-07-07T05:04:48Z","snapshot_observed_at":"2026-08-17T13:15:58.841734Z","submitted_at":"2026-07-07T05:04:48Z","title":"Beyond Refusal: A Same-Lineage Study of Aligned and Abliterated LLMs for Vulnerability Analysis","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-08T23:00:45.841744Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.05842"},"observation_digest":"sha256:d7db6b7e836a2a48d5689de5c320db4482f3a126fe6180b77670cf08694d189e","observation_id":"207ba283-9daa-4759-b969-cd7ae44ab2bb","resolution":{"observed_at":"2026-07-08T23:05:44.269506Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2607.07918","last_updated":"2026-08-14T18:25:59Z","snapshot_observed_at":"2026-08-20T23:09:41.224583Z","submitted_at":"2026-07-08T21:03:27Z","title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-07-10T15:26:23.290009Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.07918"},"observation_digest":"sha256:4248962cdd659560ccbe3d882f07cd2bf3e910f22d37c5d1dc30806c18ad0719","observation_id":"7dfadebb-76e8-40ed-b95f-16bce40796b8","resolution":{"observed_at":"2026-07-10T15:27:20.069606Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-02T07:09:30.211340Z","title":"10 Edoardo Debenedetti, Ilia Shumailov, Tianqi Fan, Jamie Hayes, Nicholas Carlini, Daniel Fabian, Christoph Kern, Chongyang Shi, Andreas Terzis, and Florian Tramèr","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13075","last_updated":"2026-07-12T20:37:19Z","snapshot_observed_at":"2026-08-17T19:58:42.196049Z","submitted_at":"2026-07-12T20:37:19Z","title":"The Entanglement Wall: Activation-Space Probes as Risk Detectors, Not Context Adjudicators","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T07:09:30.211340Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.13075"},"observation_digest":"sha256:435a8440b0e43c314c3a9c22a2ac73c6008e203758fc9dbedad1b13a64ea87f7","observation_id":"7878eb1a-ca62-4cfa-b872-3e92ccb11110","resolution":{"observed_at":"2026-08-02T07:09:30.211340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-02T07:01:19.315752Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13083","last_updated":"2026-07-13T10:14:15Z","snapshot_observed_at":"2026-08-14T17:46:33.969066Z","submitted_at":"2026-07-13T10:14:15Z","title":"Phantom Guardrails: When Self-Improving Agent Harnesses Fix Failures That Never Happened","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T07:01:19.315752Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.13083"},"observation_digest":"sha256:bfc469307385c801e1b548ad61f3e1549f742bf0990aef2b06c2b9c5f8731f41","observation_id":"af5421f9-5c79-4bd3-909e-28d7a610d778","resolution":{"observed_at":"2026-08-02T07:01:19.315752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-02T02:00:44.107363Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models, July 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14479","last_updated":"2026-07-16T01:52:54Z","snapshot_observed_at":"2026-08-18T10:17:52.636854Z","submitted_at":"2026-07-16T01:52:54Z","title":"BioTIER: A Refusal Benchmark for Targeted Biological Risk Mitigation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T02:00:44.107363Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.14479"},"observation_digest":"sha256:0e863bab9e7b1429d9bbd7ba39eaa11dad19ca14323a843e86dc4c0f3e16d9c1","observation_id":"0563645b-47ff-484d-8501-2a3c15a323ca","resolution":{"observed_at":"2026-08-02T02:00:44.107363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-01T12:58:37.815384Z","title":"2405.20947 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19262","last_updated":"2026-07-21T16:33:57Z","snapshot_observed_at":"2026-08-16T15:02:15.996715Z","submitted_at":"2026-07-21T16:33:57Z","title":"BioSecBench-Surveillance: A Verifiable Benchmark for AI Agents in Pathogen Genomic Surveillance","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-01T12:58:37.815384Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.19262"},"observation_digest":"sha256:f30250f6c83207f1a94c81b058b51fb83bb2a97b1f0b6521c713f8e9175279d4","observation_id":"f3c78e28-65a4-4ac9-bba3-b905b48479f0","resolution":{"observed_at":"2026-08-01T12:58:37.815384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-02T13:17:49.660643Z","title":"Or-bench: An over-refusal benchmark for large language models.arXiv preprint arXiv:2405.20947, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.20472","last_updated":"2026-05-24T05:49:14Z","snapshot_observed_at":"2026-08-18T04:16:12.355769Z","submitted_at":"2026-05-24T05:49:14Z","title":"Robust Critics: Defending LLMs Against Multi-Turn Attacks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T13:17:49.660643Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.20472"},"observation_digest":"sha256:49b6dcc6f1ead7419201666ec17ca6b50aa2027c2c6ae1f2c67f23c6fa53ff59","observation_id":"f0ff384a-1a37-40d2-8b34-08f71e9c1d02","resolution":{"observed_at":"2026-08-02T13:17:49.660643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-01T07:28:30.527305Z","title":"arXiv preprint arXiv:2405.20947 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21453","last_updated":"2026-07-24T02:14:44Z","snapshot_observed_at":"2026-08-19T15:31:20.809514Z","submitted_at":"2026-07-23T15:55:29Z","title":"Test-Time Scaling via Error Localization","version":2},"reference_index":124,"source":"arxiv_source","source_observed_at":"2026-08-01T07:28:30.527305Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.21453"},"observation_digest":"sha256:9906ac3b754c099ddfb71f69c12b6dbaad5bd660428f66846385fa30f0aabed6","observation_id":"0ff948cf-d26d-4f60-a12e-a822a37046d2","resolution":{"observed_at":"2026-08-01T07:28:30.527305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-07-31T16:06:04.281114Z","title":"Or-bench: An over-refusal benchmark for large language models.arXiv preprint arXiv:2405.20947, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24392","last_updated":"2026-07-27T13:07:52Z","snapshot_observed_at":"2026-08-19T19:57:31.060700Z","submitted_at":"2026-07-27T13:07:52Z","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-31T16:06:04.281114Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.24392"},"observation_digest":"sha256:11985d1689f982be09d7cb93d20bc685e60e7ed60ae988c49418896352b5c40b","observation_id":"75a6e83e-7179-488d-8857-cf1a5ded795f","resolution":{"observed_at":"2026-07-31T16:06:04.281114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-02T11:50:16.270164Z","title":"arXiv preprint arXiv:2405.20947 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24771","last_updated":"2026-06-10T13:33:10Z","snapshot_observed_at":"2026-08-21T01:49:57.357837Z","submitted_at":"2026-06-10T13:33:10Z","title":"RoCo-ACE: Rollout-Conditioned Online Distillation for Retention-Aware Knowledge Injection","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-02T11:50:16.270164Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.24771"},"observation_digest":"sha256:a30fb35d27e84bb255e7954dabe4b36e280838c11760240ad6389f1f884ee59c","observation_id":"3a627b25-8f8e-4132-9e5c-bf5bdcaa04f8","resolution":{"observed_at":"2026-08-02T11:50:16.270164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-04T01:23:34.580270Z","title":"Or-bench: Anover-refusalbenchmarkforlargelanguage models.arXiv preprint arXiv:2405.20947, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.27637","last_updated":"2026-08-01T04:59:28Z","snapshot_observed_at":"2026-08-16T19:21:02.652362Z","submitted_at":"2026-07-30T03:46:43Z","title":"MMOOC: A Comprehensive Benchmark for Out-of-Context Evaluation in Multimodal Large Language Models","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T01:23:34.580270Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.27637"},"observation_digest":"sha256:4e5b5a62db06344a1546638c06792d685566bd16a78dfb918062ac0b602b24a8","observation_id":"0514d179-cc17-43db-a904-b8f65dc95d1f","resolution":{"observed_at":"2026-08-04T01:23:34.580270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-03T00:45:59.446555Z","title":"2405.20947 , archivePrefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28685","last_updated":"2026-07-30T03:45:10Z","snapshot_observed_at":"2026-08-12T20:35:16.784837Z","submitted_at":"2026-07-30T03:45:10Z","title":"Safety, or Just Capability? A Validity Audit of Agent-Safety Benchmarks","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-03T00:45:59.446555Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2607.28685"},"observation_digest":"sha256:dc84dcc99957e6f7fdd3a29dba2530a38e77c6821e83a22079042a1f456d8edd","observation_id":"e455945e-cebc-4316-9511-2d374ae73313","resolution":{"observed_at":"2026-08-03T00:45:59.446555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T00:48:47.873645Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02665","last_updated":"2026-08-01T22:28:31Z","snapshot_observed_at":"2026-08-14T18:08:52.787563Z","submitted_at":"2026-08-01T22:28:31Z","title":"Single Canonical Prompts Underestimate LLM Safety's Surface-Form Sensitivity","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T00:48:47.873645Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2608.02665"},"observation_digest":"sha256:a270f418ce82f1b50748b941054edd46405130252b2c253cfbba9e86b1be8475","observation_id":"58f0fcc2-0fb7-4eac-9743-37fcf82efc48","resolution":{"observed_at":"2026-08-05T00:48:47.873645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-06T05:39:49.404749Z","title":"Cui,J.;Chiang,W.-L.;Stoica,I.;andHsieh,C.-J.2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05086","last_updated":"2026-08-05T17:25:27Z","snapshot_observed_at":"2026-08-18T01:11:46.539742Z","submitted_at":"2026-08-05T17:25:27Z","title":"Item Response Theory for AI Safety","version":1},"reference_index":1981,"source":"pdf_text","source_observed_at":"2026-08-06T05:39:49.404749Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2608.05086"},"observation_digest":"sha256:e13b317f05a2512a9c55e454a68dea07cc242aab5b07a221fdfc234a20be57fe","observation_id":"1bce99af-8df6-4e4b-99a9-dc0a3aba4e36","resolution":{"observed_at":"2026-08-06T05:39:49.404749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-10T18:49:33.723144Z","title":"OR-Bench: An over-refusal benchmark for large language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.06898","last_updated":"2026-08-07T07:33:25Z","snapshot_observed_at":"2026-08-20T10:58:22.320666Z","submitted_at":"2026-08-07T07:33:25Z","title":"How Should I Pick a Foundation Model for My Robot? In Favor of a Community Evaluation Framework for Social Robots","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T18:49:33.723144Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2608.06898"},"observation_digest":"sha256:6e5d84cb8fc7ee1a989bb9bbf9586095e207e7373f59133d07546742e4aabae9","observation_id":"c4e08202-2970-4b86-b433-629d5dcce9ea","resolution":{"observed_at":"2026-08-10T18:49:33.723144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-14T14:12:28.356256Z","title":"arXiv preprint arXiv:2405.20947 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13304","last_updated":"2026-08-13T14:33:15Z","snapshot_observed_at":"2026-08-17T04:46:01.333009Z","submitted_at":"2026-08-13T14:33:15Z","title":"Refusing Intent, Not Form: Wrapper-Based Intent-Group Supervision for LLM Safety","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-14T14:12:28.356256Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2608.13304"},"observation_digest":"sha256:c54cbf801eb27f5b69761be8d9da86898ae420da5ed163a609fe1633ea24047f","observation_id":"1732f4de-e3f7-48e6-a487-20841a7b14d7","resolution":{"observed_at":"2026-08-14T14:12:28.356256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2405.20947/citation-record","integrity":"/paper/2405.20947/integrity","json":"/paper/2405.20947/citation-record.json","paper":"/paper/2405.20947"},"outbound":[],"paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","latest_version":5,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-19T07:28:13.224791Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 84 inbound Pith citation observations for arXiv:2405.20947."}