{"as_of":"2026-08-18T15:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:76a335d697c205ec333ad633afacb0caa822a2a5cf12bf40294cbc60dab0d7b9","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-22T01:19:00.268857Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T08:40:01.438522Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.21545","snapshot_observed_at":"2026-07-11T16:42:49.284801Z","title":"Refusalbench: Why refusal rate misranks frontier llms on biological research prompts.arXiv preprint arXiv:2605.21545, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.05462","last_updated":"2026-07-21T17:05:39Z","snapshot_observed_at":"2026-08-18T12:55:08.383819Z","submitted_at":"2026-07-06T01:46:07Z","title":"BioSecBench-Refusal: A paired metric for performance and alignment in agentic biosecurity risk assessment","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-11T16:42:49.284801Z"},"links":{"cited_paper":"/paper/2605.21545","citing_paper":"/paper/2607.05462"},"observation_digest":"sha256:d8eb0a8c9be285ab4c619685f61a978012861d1e3f12926687f38de0fb8f4040","observation_id":"910685a1-f3f7-4069-97bd-72855cabab83","resolution":{"observed_at":"2026-07-11T16:42:49.284801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.21545","snapshot_observed_at":"2026-08-02T08:40:01.438522Z","title":"Refusalbench: Why refusal rate misranks frontier llms on biological research prompts.arXiv preprint arXiv:2605.21545, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.05462","last_updated":"2026-07-21T17:05:39Z","snapshot_observed_at":"2026-08-18T12:55:08.383819Z","submitted_at":"2026-07-06T01:46:07Z","title":"BioSecBench-Refusal: A paired metric for performance and alignment in agentic biosecurity risk assessment","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T08:40:01.438522Z"},"links":{"cited_paper":"/paper/2605.21545","citing_paper":"/paper/2607.05462"},"observation_digest":"sha256:8bd94b9d51cbb030f1290095652746de57447f05a9b0c4948d785c2ed9ffd4ae","observation_id":"9e8cf392-3063-4472-94f6-bd8b27376ef6","resolution":{"observed_at":"2026-08-02T08:40:01.438522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.21545/citation-record","integrity":"/paper/2605.21545/integrity","json":"/paper/2605.21545/citation-record.json","paper":"/paper/2605.21545"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s41586-025-09429-6","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"One-shot design of functional protein binders with BindCraft","venue":"Nature","work_id":"780d769e-46eb-4fb5-b787-624ca3668fda","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:4d6b82a4f27207690994e24fe1c312c2e624beca946fc7752fc2291c74fe28f1","observation_id":"ca587e6c-8cdc-440a-8aeb-92d5ed1b749a","resolution":{"observed_at":"2026-05-22T01:20:51.616120Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ProteinCrow: A language model agent that can design proteins","venue":null,"work_id":"623814e0-7b90-45bf-b58c-6e49a8fcdfc9","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:63f90b33c60c458a16d350720aa97230c92538b95303c35a3d32ed58b9fdb1d2","observation_id":"87ea9d68-9a98-481e-931a-9a192853a492","resolution":{"observed_at":"2026-05-22T01:20:52.359382Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1f1a84dd-b60a-4f6e-8110-e3dfc1e229cb","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:fe3bafe8420b369b4a1f36e63f0e33b015cf672940b9fa6b9b586e90387dd91b","observation_id":"71322988-55b6-48c7-8f07-74e2ec7021db","resolution":{"observed_at":"2026-05-22T01:20:52.362821Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Beyond protein language models: An agentic LLM framework for mechanistic enzyme design","venue":null,"work_id":"9bc6710d-f037-4c66-8d73-44b8a5bd35b6","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:1f03f7f83dcdafc7a87305ff55c4ace5910353bb13a4c793edbb1c22ddd5d248","observation_id":"c5d25555-0c66-4abb-b731-6156427807f6","resolution":{"observed_at":"2026-05-22T01:20:52.366248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.19423","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T20:56:14.410913Z","title":"arXiv:2511.19423","venue":null,"work_id":"39814443-7b0f-49f2-a14a-78a678298636","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:408d097e49248fb49308ca78c5235dd2b567011efe84dc8a389c27e6ef6e387e","observation_id":"8cdd4475-adb8-433d-bf8c-f5138fa06e56","resolution":{"observed_at":"2026-05-22T01:20:51.962759Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ProtoCy- cle: Reflective tool-augmented planning for text-guided protein design","venue":null,"work_id":"50c7da93-f432-430c-8b83-144454088568","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:1a1f063170e33b2a09fb26d324a4741b44598c6fa72561c198f72ebef6395ba4","observation_id":"84ecc782-35d6-44ae-803c-60d44eb0e743","resolution":{"observed_at":"2026-05-22T01:20:52.337718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.16896","last_updated":"2026-04-18T08:09:10Z","snapshot_observed_at":"2026-08-15T17:39:18.143896Z","submitted_at":"2026-04-18T08:09:10Z","title":"ProtoCycle: Reflective Tool-Augmented Planning for Text-Guided Protein Design","version":1},"cited_work":{"arxiv_id":"2604.16896","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16896","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ProtoCycle: Reflective Tool-Augmented Planning for Text-Guided Protein Design","venue":"q-bio.QM","work_id":"68a684f7-a4c8-4998-b72a-c394d9db0f6a","year":2026},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2604.16896","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:c3819fcc547107d84ac0f6bb3e8f4bead18a250810ee9d993918677b67dd3dfc","observation_id":"8c811739-6118-49d1-adad-19f4f0a926db","resolution":{"observed_at":"2026-05-22T01:20:51.957519Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1002/pro.70547","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ProteinMCP: An agentic AI framework for autonomous protein engineer- ing","venue":"Protein Science","work_id":"9af4cd67-679f-45ff-b503-cb2e54743934","year":2026},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:c811092a61c4dfe9efb6288206af60c95de008e994d69fdcdcbc18b21ada500d","observation_id":"1ec75940-ec6f-40b1-84a5-add990c4bbc5","resolution":{"observed_at":"2026-05-22T01:20:51.596971Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Using a GPT-5-driven autonomous lab to optimize the cost and titer of cell-free protein synthesis","venue":null,"work_id":"8c4c1adc-e16d-49d3-b040-1322143f3a93","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:191c9c04cbc2b4f99f62aaf597675100cd39fcdcd6156bdf6023d0c54a315b32","observation_id":"1af6843a-8940-415d-b357-703c74c64aeb","resolution":{"observed_at":"2026-05-22T01:20:52.333877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.64898/2026.02.05.703998","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"bioRxiv (Cold Spring Harbor Laboratory)","work_id":"215b6ad8-b21f-4c16-945e-716230947a21","year":2026},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:6f92d1486b1110c19a6d951ca41398ae533f0f34427444532bd0387320e54284","observation_id":"59b0a91d-7833-4bb1-b207-2e7a3cfbef34","resolution":{"observed_at":"2026-05-22T01:20:51.583014Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Agen- tic BAIM–LLM evaluation (ABLE): Bench- marking LLM use of protein design tools","venue":null,"work_id":"2ff077a7-9394-4217-9b26-38b24cc1f43f","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:65f4920a08aaec510e6d3841847e9497092fc24796b6daa067c495a2cb3e669f","observation_id":"ea3fd858-7d1d-49a2-9ad6-0e99a0684465","resolution":{"observed_at":"2026-05-22T01:20:52.297123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01263","last_updated":"2024-04-01T11:50:35Z","snapshot_observed_at":"2026-08-14T04:19:38.572552Z","submitted_at":"2023-08-02T16:30:40Z","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","version":3},"cited_work":{"arxiv_id":"2308.01263","doi":"10.48550/arxiv.2308.01263","metadata_source":"pith","pith_arxiv_id":"2308.01263","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","venue":"cs.CL","work_id":"bd953600-1547-4c1e-ade7-219a9f7cfe7a","year":2023},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2308.01263","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:5cce5fab8728ddcb24a6dfcf7ac88f356045eca6cde6717d0d56d9c1fa0515e0","observation_id":"785c5b59-3e0a-4d58-af23-a26f307823ed","resolution":{"observed_at":"2026-05-22T01:20:51.901268Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"OR-Bench: An over- refusal benchmark for large language mod- els","venue":null,"work_id":"2d0ee89a-aff7-4257-bc1f-c08580d80493","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:72da173bd00143ef5b4036f4796b14220880a56f7fb1944f9814bde5ca0f07b0","observation_id":"7c5153cb-46d4-4fde-a475-6752eb7f8bf2","resolution":{"observed_at":"2026-05-22T01:20:52.305301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20947","last_updated":"2025-06-15T21:44:25Z","snapshot_observed_at":"2026-08-16T13:47:22.319152Z","submitted_at":"2024-05-31T15:44:33Z","title":"OR-Bench: An Over-Refusal Benchmark for Large Language Models","version":5},"cited_work":{"arxiv_id":"2405.20947","doi":"10.48550/arxiv.2405.20947","metadata_source":"pith","pith_arxiv_id":"2405.20947","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OR- Bench: An over-refusal benchmark for large language models","venue":"cs.CL","work_id":"5dc76f29-8555-4005-954c-6e085345fc2f","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2405.20947","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:87e9b9d35ff563ae6b223c8a827f4bd9682d68c615ea645a240072368a8c7987","observation_id":"757b7704-4f27-4c1d-9dd3-cfca173d462b","resolution":{"observed_at":"2026-05-22T01:20:51.889010Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16974","last_updated":"2024-12-22T11:16:53Z","snapshot_observed_at":"2026-08-15T12:31:56.442643Z","submitted_at":"2024-12-22T11:16:53Z","title":"Cannot or Should Not? Automatic Analysis of Refusal Composition in IFT/RLHF Datasets and Refusal Behavior of Black-Box LLMs","version":1},"cited_work":{"arxiv_id":"2412.16974","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.16974","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cannot or should not? Automatic analysis of refusal composition in IFT/RLHF datasets and refusal behavior of black-box LLMs","venue":null,"work_id":"815e3862-0c9a-480c-9e0b-73abe7c89dd5","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2412.16974","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:df946232ded11e08896d6d4efdb7004eed1d79e5228e96154826980625450669","observation_id":"fded0e28-e891-4b7a-8aff-bed07bfe1418","resolution":{"observed_at":"2026-05-22T01:20:51.947425Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Forbid- den science: Dual-use AI challenge benchmark and scientific refusal tests","venue":null,"work_id":"0407f760-3f69-4aaf-88a8-d92dd9984f85","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:ee4c1cd41b3d9043bf3e05f7a5386268af3937549d1c79b165ea390fde9f4b06","observation_id":"33cd4966-0a11-4cb1-9c44-6384baee4b01","resolution":{"observed_at":"2026-05-22T01:20:52.330044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06867","last_updated":"2025-02-08T04:27:33Z","snapshot_observed_at":"2026-08-15T16:52:47.335780Z","submitted_at":"2025-02-08T04:27:33Z","title":"Forbidden Science: Dual-Use AI Challenge Benchmark and Scientific Refusal Tests","version":1},"cited_work":{"arxiv_id":"2502.06867","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06867","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv:2502.06867","venue":null,"work_id":"70f3b1da-bcf0-4ce4-aef9-43c20b9b0589","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2502.06867","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:2fc598185c77c2ed70a9e0f2ba76c2da59b23f1c65450fb1ccbb9dc82015b11e","observation_id":"701f45cb-5f99-4851-b73d-42b3d9100d30","resolution":{"observed_at":"2026-05-22T01:20:51.936763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Political censorship in large language models originating from China","venue":null,"work_id":"3878cd57-88ce-4c5e-bc22-d921fb1c102a","year":2026},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:1e55efeaf74c4ecb73791594b6d5790ed2881797f8edc892b792425a091b8667","observation_id":"7deeae4d-88e2-4b7c-ba1b-3b23228cd846","resolution":{"observed_at":"2026-05-22T01:20:52.313076Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Virol- ogy capabilities test (VCT): A multimodal virology Q&A benchmark","venue":null,"work_id":"ec99c177-ac03-486d-80f4-9eaac3de27d4","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:e449163c92c1a77be217d1a9aa664a532aaa19e5cc4528ba5884b940f3c23732","observation_id":"40a6b9f7-813d-4825-b7c4-f485731ec9a6","resolution":{"observed_at":"2026-05-22T01:20:52.326050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16137","last_updated":"2025-04-29T15:14:35Z","snapshot_observed_at":"2026-08-16T11:24:31.452137Z","submitted_at":"2025-04-21T21:04:01Z","title":"Virology Capabilities Test (VCT): A Multimodal Virology Q&A Benchmark","version":2},"cited_work":{"arxiv_id":"2504.16137","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.16137","snapshot_observed_at":"2026-07-01T20:56:14.444125Z","title":"arXiv [preprint]","venue":null,"work_id":"e9760385-440f-46ad-bc95-b2d5ebcd370b","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2504.16137","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:e6af159d2cf8e7e5d9e6db4d7e8b64cd670605d01f1c08d078f8f7ad31339b34","observation_id":"9a754cc8-6fba-471a-8dfe-81bb214e9066","resolution":{"observed_at":"2026-05-22T01:20:51.895253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Can large language models democratize access to dual-use biotechnology? arXiv preprint","venue":null,"work_id":"ec33819d-a5e0-4d8f-bec8-38ad170c1949","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:cb37936f83826b0113d63638780c9c0a6b8021f80c7a9cc2e73d55c3d3e20256","observation_id":"aeb64051-00e6-4474-9ae2-1a7f74900b65","resolution":{"observed_at":"2026-05-22T01:20:52.369823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03809","last_updated":"2023-06-06T15:52:05Z","snapshot_observed_at":"2026-08-16T15:26:07.109585Z","submitted_at":"2023-06-06T15:52:05Z","title":"Can large language models democratize access to dual-use biotechnology?","version":1},"cited_work":{"arxiv_id":"2306.03809","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.03809","snapshot_observed_at":"2026-07-01T19:46:10.221526Z","title":"Can JOURNAL OF LATEX CLASS FILES, VOL. 14, NO. 8, AUGUST 2021 13 large language models democratize access to dual-use biotechnology?","venue":null,"work_id":"de235a65-50bc-48ff-8468-c2e0f9a5352c","year":2021},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2306.03809","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:8a86e227f8a3120e4ca1e7eef8ab09b3d80bba13ba4d78ec55da23713161e4a6","observation_id":"9c3e08c0-dcf5-4aa6-b0e2-719652a1dbf0","resolution":{"observed_at":"2026-05-22T01:20:51.931094Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The next-generation Open Targets platform: reimag- ined, redesigned, rebuilt","venue":null,"work_id":"5250efd9-88f3-4a3d-b289-5521f0b8aa83","year":2023},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:95e3086f6730b1f7f0184de7d301ecf0fe97b26be5dd6433f4d0abf626e4da40","observation_id":"647aa87d-09ae-4956-a3ea-6f1ace9337b1","resolution":{"observed_at":"2026-05-22T01:20:52.348456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1093/nar/gkae1010","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"UniProt: the uni- versal protein knowledgebase in 2025","venue":"Nucleic Acids Research","work_id":"dbdc78a3-ee26-47ca-892c-66b69b43dbae","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:f555a906396c04f6fd3415f1d114315cc097282140d5017badbdf16205028f72","observation_id":"107763b1-d770-4f69-8057-3f1d10f4d527","resolution":{"observed_at":"2026-05-22T01:20:51.610655Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-05-20T10:53:09.155891+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T10:53:09.155891+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NVIDIA Nemotron 3 Su- per: A 120B hybrid Mamba-Transformer MoE model for agentic reasoning","venue":null,"work_id":"e3027847-9e00-4609-b9aa-c9b300a985f3","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:6fd867aa047717efde6b9319e5536f1138aa281be1cd7e97d9016a57d2e3c9e6","observation_id":"bc2c5a7a-4cc5-4e1a-aa5a-85986ca94a4e","resolution":{"observed_at":"2026-05-22T01:20:52.320830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open-weights release; 12B active / 120B total parameters; 1M-token context","venue":null,"work_id":"02081171-d51f-43d1-bad4-e0cb99488f2f","year":2026},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:2f907b86fc39ce505b41af6f9abf63ae2a34f55dde64028f45f4c1c319aa6682","observation_id":"a75acdab-0de0-420b-8df7-c04394647de6","resolution":{"observed_at":"2026-05-22T01:20:52.301203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"SORRY- Bench: Systematically evaluating large lan- guage model safety refusal","venue":null,"work_id":"595bcb4a-9fa1-4ce7-abd8-235e96af559d","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:63838bedc1d34ef5f9c6d9de770143bc9eacc776f5e40013d1e3404f11c62324","observation_id":"c6013dec-24c1-446d-9c90-83ac80a86860","resolution":{"observed_at":"2026-05-22T01:20:52.316971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14598","last_updated":"2025-03-01T21:45:36Z","snapshot_observed_at":"2026-08-17T11:35:24.473105Z","submitted_at":"2024-06-20T17:56:07Z","title":"SORRY-Bench: Systematically Evaluating Large Language Model Safety Refusal","version":2},"cited_work":{"arxiv_id":"2406.14598","doi":"10.48550/arxiv.2406.14598","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14598","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2406.14598 (2025)","venue":"arXiv (Cornell University)","work_id":"5f6c39b5-65d4-4261-a0d0-63007ebce626","year":2025},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2406.14598","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:1f6522fc5c2d3f8ec74cabe9a57eb0e1fa1660371394222608c9830555d12a60","observation_id":"6b539532-c25e-4445-ad80-74ef7a44f0c4","resolution":{"observed_at":"2026-05-22T01:20:51.907305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Content Analysis: An Introduction to Its Methodology","venue":null,"work_id":"fab8fb5b-cb48-4d75-be4e-c01c574eb3be","year":2004},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:9dc3efcf231d4760bb526ce3f93b4141c153b51c527a570fe90749e278aa8a16","observation_id":"ffa28ec7-3751-42ba-a964-4e66f42c28d6","resolution":{"observed_at":"2026-05-22T01:20:52.352023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The art of saying no: Contextual non- compliance in language models","venue":null,"work_id":"a5ce122e-59e7-401e-8b20-819489c068d6","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:174f40e4719369c6503bb88150026a84070470009af45c6bd7bf0fe7db82b39b","observation_id":"00fff7df-d088-4920-ab28-8219d46a02cc","resolution":{"observed_at":"2026-05-22T01:20:52.344699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12043","last_updated":"2024-11-22T17:48:57Z","snapshot_observed_at":"2026-08-17T12:38:00.773511Z","submitted_at":"2024-07-02T07:12:51Z","title":"The Art of Saying No: Contextual Noncompliance in Language Models","version":2},"cited_work":{"arxiv_id":"2407.12043","doi":"10.48550/arxiv.2407.12043","metadata_source":"pith","pith_arxiv_id":"2407.12043","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The art of saying no: Contex- tual noncompliance in language models","venue":"cs.CL","work_id":"fb17e5aa-5a94-41a8-bfde-cea56437a988","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2407.12043","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:db6a82e54ee0653d56ccc66ef8882e6cd54f3f84d3bda44600a06e9fef40312e","observation_id":"1b504a5b-3776-4a91-9246-4843a4b2df69","resolution":{"observed_at":"2026-05-22T01:20:51.941888Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-16T03:49:00.703994Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":"2212.08073","doi":"10.48550/arxiv.2212.08073","metadata_source":"pith","pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Constitutional AI: Harmlessness from AI Feedback","venue":"cs.CL","work_id":"faaaa4e0-2676-4fac-a0b4-99aef10d2095","year":2022},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:9e88726311a5a29fcd7cdee0585b805818304b24b9c4b0dcd0a15200ee72e712","observation_id":"b5d7b3ba-568f-4d42-aa10-7fa20ca245b4","resolution":{"observed_at":"2026-05-22T01:20:51.919483Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-15T03:08:13.900899+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-15T03:08:13.900899+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Anthropic’s responsible scaling policy","venue":null,"work_id":"121a68c2-fe7b-41ec-be75-5e7f615e2db0","year":2026},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:3049382d530819c25f42aa4378ecac93b1c231540dda6d5539f5333db67a7e8f","observation_id":"cd4dd58d-1202-454f-a890-7ae279186d4c","resolution":{"observed_at":"2026-05-22T01:20:52.341138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1126/science","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A., Mathur, S., Salabert, D., Ballot, J., R´egulo, C., Metcalfe, T","venue":"Science","work_id":"656f6b47-9c8d-489d-af8f-5d0f17981423","year":2019},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:7b40e98fa4b4ffa622483628188d132936e2ad7d3fc18f24e118752601a110cc","observation_id":"6d23fe70-0226-4931-91fe-1975d52042fc","resolution":{"observed_at":"2026-05-22T01:20:51.604548Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dual use of ar- tificial intelligence-powered drug discovery","venue":null,"work_id":"bd386080-3393-4074-b802-6ba40fccfdcd","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:70420092a4f40b88d7bcf0334b7b44389e327817ccc2bd39f22fa467fb6c8139","observation_id":"da7d68e0-3579-4211-b94c-d698de212477","resolution":{"observed_at":"2026-05-22T01:20:52.373431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1928e755-32ab-48cb-a7c1-a1e9305c601c","year":null},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:41b1bb3990147e9b07ff7dccebc6a6d52bfb376c313aa6c808635de4a57f8611","observation_id":"4bce7a55-a6b8-43ed-814e-3a9912cdd7a3","resolution":{"observed_at":"2026-05-22T01:20:52.355717Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Training language models to follow instructions with hu- man feedback","venue":null,"work_id":"1bed68df-0ac3-4f35-94ba-41e76b008370","year":2022},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:fffe8b299789d69d2be1f7bafc47c55a02d33af93a44fbe8ca780e08223eca72","observation_id":"a5dcc4fd-ec7f-48e4-9b1a-0b0cac51579a","resolution":{"observed_at":"2026-05-22T01:20:52.309141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":"2203.02155","doi":"10.1007/s00354-022-00198-8","metadata_source":"pith","pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Training language models to follow instructions with human feedback","venue":"cs.CL","work_id":"52aff42f-4fa9-4fcf-bdb3-1459b9bebf65","year":2022},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:487e5d3140c4a6c428f7665c26adaaee50376d41badb790b983358bb86b26b27","observation_id":"a9b5081e-678c-4a61-bda6-d4b89781c5c7","resolution":{"observed_at":"2026-05-22T01:20:51.952881Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Harm- Bench: A standardized evaluation framework for automated red teaming and robust re- fusal","venue":null,"work_id":"3d8547a8-942d-4869-a997-f551985e57ea","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:706995cd84eab72aad88b58b20685768f80bb1352a659293d6c12d861a9e5211","observation_id":"a9a33a66-c8c8-4fb4-be63-55d036d2914b","resolution":{"observed_at":"2026-05-22T01:20:52.376614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04249","last_updated":"2024-02-27T04:43:08Z","snapshot_observed_at":"2026-08-16T09:07:20.265665Z","submitted_at":"2024-02-06T18:59:08Z","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","version":2},"cited_work":{"arxiv_id":"2402.04249","doi":"10.48550/arxiv.2402.04249","metadata_source":"pith","pith_arxiv_id":"2402.04249","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","venue":"cs.LG","work_id":"b0b0303f-2444-4789-a979-8153624312ff","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2402.04249","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:90db648715b5f7d7a1fec5944e6324c7e32c404d1a3754faed704bf1a99b93c2","observation_id":"07c00f47-c693-4eea-96ce-da6e2eed708c","resolution":{"observed_at":"2026-05-22T01:20:51.925495Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03218","last_updated":"2024-05-15T19:16:09Z","snapshot_observed_at":"2026-08-15T06:18:34.739211Z","submitted_at":"2024-03-05T18:59:35Z","title":"The WMDP Benchmark: Measuring and Reducing Malicious Use With Unlearning","version":7},"cited_work":{"arxiv_id":"2403.03218","doi":"10.48550/arxiv.2403.03218","metadata_source":"pith","pith_arxiv_id":"2403.03218","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The WMDP Benchmark: Measuring and Reducing Malicious Use With Unlearning","venue":"cs.LG","work_id":"d05f8523-8089-4fdb-9c07-463952166528","year":2024},"citing_paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-22T01:19:00.268857Z"},"links":{"cited_paper":"/paper/2403.03218","citing_paper":"/paper/2605.21545"},"observation_digest":"sha256:9027315d94b33d18a1aa62283be723bfff0c1a491118274532c0b7b693980f11","observation_id":"a356293f-7cbf-43d2-9107-6d4e18597a37","resolution":{"observed_at":"2026-05-22T01:20:51.913107Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-05-22T21:23:22.693969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T21:23:22.693969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.21545","last_updated":"2026-05-20T09:53:31Z","latest_version":1,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-13T02:25:36.321787Z","submitted_at":"2026-05-20T09:53:31Z","title":"RefusalBench: Why Refusal Rate Misranks Frontier LLMs on Biological Research Prompts"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":3,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":2,"verified_exact":17,"verified_fuzzy":19},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 2 inbound Pith citation observations for arXiv:2605.21545."}