{"as_of":"2026-08-08T13:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:af9ddbaf1784958490aa51af01de1b3f58fd3c18d58b160be3af792565dd5869","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:52:31.262024Z","state":"measured"},{"denominator":58,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":58,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.17332/citation-record","integrity":"/paper/2505.17332/integrity","json":"/paper/2505.17332/citation-record.json","paper":"/paper/2505.17332"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:26.322302Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.322302Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5b9775345fe901f27890031fa09c0f545a0a11c7f286fb72ae2f600f7b2242b8","observation_id":"26b1aae7-672a-45e4-b9c2-5039e9325947","resolution":{"observed_at":"2026-08-07T14:52:26.322302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:26.464178Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.464178Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:1b30b0be14316addaa214a223614780e7a81eec1adb67c5598eabfbb640f5845","observation_id":"7c087a5a-b0ad-4bcf-b766-8d2e8440c8da","resolution":{"observed_at":"2026-08-07T14:52:26.464178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-07T14:52:26.615041Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.615041Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:cde200ff319d9552295b0bd26f59fe062937b2fd67eadbff55894cad264d205d","observation_id":"64a2ae5a-527d-47ca-b002-dc740e031058","resolution":{"observed_at":"2026-08-07T14:52:26.615041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.997264Z","title":null,"venue":null,"work_id":"dffecbf0-cf44-442b-8592-945b6ee96239","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.750767Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5c07477db2db74e08ed2143e616c0db57e4ec044ea2c64a6409d8c364355d114","observation_id":"c2631a25-09ad-4c04-a86b-ff246c00d569","resolution":{"observed_at":"2026-08-07T14:52:33.061472Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19794","last_updated":"2025-06-11T16:24:02Z","snapshot_observed_at":"2026-08-07T22:59:48.143630Z","submitted_at":"2024-12-27T18:47:05Z","title":"MVTamperBench: Evaluating Robustness of Vision-Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19794","snapshot_observed_at":"2026-08-07T14:52:26.868304Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.868304Z"},"links":{"cited_paper":"/paper/2412.19794","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:2828fba62db11da698dd5494d24c64e75b96a6dae439a46ebc6246e2c0f84fab","observation_id":"d0965ad8-1162-4095-8b4a-402b8dc9755b","resolution":{"observed_at":"2026-08-07T14:52:26.868304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.861880Z","title":null,"venue":null,"work_id":"411e2849-32f3-4a29-a6ea-5c554f38bf91","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:26.976950Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:9e3c97cea6d9bb57f065ec9e7bc57ef8618eba40e578d2411b10eba1a723b576","observation_id":"e895a1df-56ae-4bd4-a19c-9c5d42c6c30f","resolution":{"observed_at":"2026-08-07T14:52:32.928539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.727447Z","title":null,"venue":null,"work_id":"abae5364-e8ba-4733-9040-65987fbd49e2","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.115403Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5b1a0c7735a078396eeb4ae01441c124f927f59b979c1c81a550a7e66d6f2e45","observation_id":"eb97f4e2-8ef0-4ace-883d-09589e055212","resolution":{"observed_at":"2026-08-07T14:52:32.801949Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:27.165158Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.165158Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:6260f2766df252568a05cb1e709e9c65123fc011a74fec3bf32ca6c29f7173f4","observation_id":"550c9ee5-1b49-4834-93b7-21747f8a4949","resolution":{"observed_at":"2026-08-07T14:52:27.165158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03590","last_updated":"2024-11-27T21:15:02Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-11-27T21:15:02Z","title":"Enhancing Document AI Data Generation Through Graph-Based Synthetic Layouts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03590","snapshot_observed_at":"2026-08-07T14:52:27.217122Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.217122Z"},"links":{"cited_paper":"/paper/2412.03590","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:aa0e6d343bc40508f9b4b762bd3a5e69e1feda7dbf15d637196627682e486527","observation_id":"6ee05890-3ee7-49f1-ac40-767856edd6ea","resolution":{"observed_at":"2026-08-07T14:52:27.217122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:27.279376Z","title":"Do, Yan Xu, and Pascale Fung","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.279376Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:c1b2170557e45c36c584c18eec6e7e6dc1c318ba7256d279392a8700d366d360","observation_id":"207e7e05-166f-44cc-b4a5-d9f29d35b8d7","resolution":{"observed_at":"2026-08-07T14:52:27.279376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-08-07T14:52:27.438818Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.438818Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:44de4c4f983852c8628f609f1c88c8fcf9ed67c24f0cca206768d3a651d17790","observation_id":"30a523f3-53cd-48df-9482-3c14107d1c63","resolution":{"observed_at":"2026-08-07T14:52:27.438818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.01318","last_updated":"2024-10-31T22:26:40Z","snapshot_observed_at":"2026-08-02T14:59:12.115203Z","submitted_at":"2024-03-28T02:44:02Z","title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.01318","snapshot_observed_at":"2026-08-07T14:52:27.510965Z","title":"Pappas, Florian Tramer, Hamed Hassani, and Eric Wong","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.510965Z"},"links":{"cited_paper":"/paper/2404.01318","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:881332ce731a2e9aecc57d77e70af3e7cf115b7cf81018918f9dfe339639f89a","observation_id":"a56ae450-2731-4a80-a54a-990f265db275","resolution":{"observed_at":"2026-08-07T14:52:27.510965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16135","last_updated":"2025-03-04T07:00:10Z","snapshot_observed_at":"2026-08-06T00:51:21.827894Z","submitted_at":"2024-06-23T15:15:17Z","title":"Crosslingual Capabilities and Knowledge Barriers in Multilingual Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16135","snapshot_observed_at":"2026-08-07T14:52:27.565303Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.565303Z"},"links":{"cited_paper":"/paper/2406.16135","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:9a8d092feed284718f92c48780cd2e6f1556625996d80f59a6a86ab39828819d","observation_id":"f7d9ba58-cc46-45fd-a320-59388953a377","resolution":{"observed_at":"2026-08-07T14:52:27.565303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:27.646088Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.646088Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:9e2abc613fe0b0f7fb8eba20af570cfaea46a4e1ccbac3534855c88c16b85aa8","observation_id":"7f04b227-b7e4-436e-b7d4-918c16af380e","resolution":{"observed_at":"2026-08-07T14:52:27.646088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T14:52:27.742920Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.742920Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:2e8859287d8a3552311b402c3ccc98c298c5e341b7092daad8dc728e5b921f0e","observation_id":"a93f0125-3504-4efb-a012-be1992f6f87d","resolution":{"observed_at":"2026-08-07T14:52:27.742920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10602","last_updated":"2025-04-25T10:53:27Z","snapshot_observed_at":"2026-07-06T18:31:27.512384Z","submitted_at":"2024-06-15T11:31:39Z","title":"Multilingual Large Language Models and Curse of Multilinguality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10602","snapshot_observed_at":"2026-08-07T14:52:27.834797Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.834797Z"},"links":{"cited_paper":"/paper/2406.10602","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:d7e8fd7f22486d9033175bde5db15006b3c90998414022b5113d3b5a5ed8b031","observation_id":"4354f0a9-961f-4be5-b27d-78afba2db335","resolution":{"observed_at":"2026-08-07T14:52:27.834797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.09509","last_updated":"2022-07-14T13:04:29Z","snapshot_observed_at":"2026-07-06T12:49:18.486644Z","submitted_at":"2022-03-17T17:57:56Z","title":"ToxiGen: A Large-Scale Machine-Generated Dataset for Adversarial and Implicit Hate Speech Detection","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.09509","snapshot_observed_at":"2026-08-07T14:52:27.948003Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:27.948003Z"},"links":{"cited_paper":"/paper/2203.09509","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:0036ed2858f6a4f4d91d7bfcb0682fe7ef691a60992f37d178b22a79f66ce8ea","observation_id":"556dabd3-900e-4d59-a3f7-865468e7a661","resolution":{"observed_at":"2026-08-07T14:52:27.948003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-03T06:34:42.243765Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T14:52:28.092591Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.092591Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:19df09de2b986d0d3e7d03166ee6aab148481800bd1283446d617943b33b07e4","observation_id":"fe891c49-9c68-4b27-a31d-636d3040cd64","resolution":{"observed_at":"2026-08-07T14:52:28.092591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-07T14:52:28.202392Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.202392Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:e52ccbbb2b75a97d0a474a861fdfeb62d29744b9de6d72950b64555f6b07ff4b","observation_id":"289d9c3d-4374-49f1-90c9-abe2b0413ec1","resolution":{"observed_at":"2026-08-07T14:52:28.202392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.00515","last_updated":"2024-11-10T22:02:27Z","snapshot_observed_at":"2026-07-29T20:40:25.374189Z","submitted_at":"2024-06-01T17:48:15Z","title":"A Survey on Large Language Models for Code Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.00515","snapshot_observed_at":"2026-08-07T14:52:28.304469Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.304469Z"},"links":{"cited_paper":"/paper/2406.00515","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:612874ffe26e280e6ea8ea789d3e6265e91626f1c441bc81702de8fdef68a663","observation_id":"e14fc6fb-c632-433d-b421-a9306e3bc01b","resolution":{"observed_at":"2026-08-07T14:52:28.304469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:28.406960Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.406960Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:685d2898541d1efe5613a45e38b4d3b4ed1a1bac6dee05156d4d99c153ed65d5","observation_id":"bb0d6a1a-e472-4331-a9bb-dd6f620818a6","resolution":{"observed_at":"2026-08-07T14:52:28.406960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04392","last_updated":"2024-09-09T06:25:33Z","snapshot_observed_at":"2026-08-04T07:01:53.971515Z","submitted_at":"2024-04-05T20:31:45Z","title":"Fine-Tuning, Quantization, and LLMs: Navigating Unintended Outcomes","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04392","snapshot_observed_at":"2026-08-07T14:52:28.533469Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.533469Z"},"links":{"cited_paper":"/paper/2404.04392","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:75c9f8104dc293ee39523d64b6f61e36fe961d237305be9c5a7f784a01d80d19","observation_id":"d95e5fad-f1cd-4af4-9819-486e74c05bc6","resolution":{"observed_at":"2026-08-07T14:52:28.533469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05044","last_updated":"2024-06-07T12:05:46Z","snapshot_observed_at":"2026-08-03T21:40:53.683032Z","submitted_at":"2024-02-07T17:33:54Z","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05044","snapshot_observed_at":"2026-08-07T14:52:28.676227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.676227Z"},"links":{"cited_paper":"/paper/2402.05044","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:06e8ada90273cfd60bce88cc2054fdf9acc19e1426662d60b4df9f04b788ef56","observation_id":"7e09b11c-0844-42b0-9106-664da1d3e6ec","resolution":{"observed_at":"2026-08-07T14:52:28.676227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12599","last_updated":"2024-08-22T17:59:04Z","snapshot_observed_at":"2026-07-06T19:04:43.716629Z","submitted_at":"2024-08-22T17:59:04Z","title":"Controllable Text Generation for Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12599","snapshot_observed_at":"2026-08-07T14:52:28.795914Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.795914Z"},"links":{"cited_paper":"/paper/2408.12599","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:18723231fb428e7a658e8e566fee0e21f558f9c521b7febe87727f783a9d7ab2","observation_id":"3ffc685f-ab4a-4baa-923c-52ef5ab7142b","resolution":{"observed_at":"2026-08-07T14:52:28.795914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T14:52:28.909375Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.909375Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:82036c0af1fa3c865373cd86edc577e533465fdaa308c3a67c82014d9fdf63ab","observation_id":"e773cb32-b65a-4db0-8695-1cd4132b6e08","resolution":{"observed_at":"2026-08-07T14:52:28.909375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04656","last_updated":"2024-05-07T20:29:48Z","snapshot_observed_at":"2026-07-06T18:11:13.757248Z","submitted_at":"2024-05-07T20:29:48Z","title":"Corporate Communication Companion (CCC): An LLM-empowered Writing Assistant for Workplace Social Media","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04656","snapshot_observed_at":"2026-08-07T14:52:28.995452Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.995452Z"},"links":{"cited_paper":"/paper/2405.04656","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:55e95e778b3f90428a958ed06c8cd40929addd6dc7434268c79195f866718875","observation_id":"8ada4d7c-2f56-49aa-8e40-16cf10badd7a","resolution":{"observed_at":"2026-08-07T14:52:28.995452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.01181","last_updated":"2024-04-02T01:56:56Z","snapshot_observed_at":"2026-07-06T15:22:09.036511Z","submitted_at":"2023-05-02T03:27:27Z","title":"A Paradigm Shift: The Future of Machine Translation Lies with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.01181","snapshot_observed_at":"2026-08-07T14:52:29.088283Z","title":"Wong, Siyou Liu, and Longyue Wang","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.088283Z"},"links":{"cited_paper":"/paper/2305.01181","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:95187afee95c52921894cb725e0ad9dda0822efd2d3348436047c1324ba484a1","observation_id":"da7635fc-2190-4561-b9c3-f2daa71a078a","resolution":{"observed_at":"2026-08-07T14:52:29.088283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04249","last_updated":"2024-02-27T04:43:08Z","snapshot_observed_at":"2026-07-06T17:26:23.067923Z","submitted_at":"2024-02-06T18:59:08Z","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04249","snapshot_observed_at":"2026-08-07T14:52:29.187055Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.187055Z"},"links":{"cited_paper":"/paper/2402.04249","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:ec721ef37e66ccb8df6c4bba9f63e89398bbfd131c18d8cdaa6d7176b6c52567","observation_id":"e70cb713-e48b-4f6a-8c0b-d533273190ac","resolution":{"observed_at":"2026-08-07T14:52:29.187055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.583945Z","title":null,"venue":null,"work_id":"10833ee9-6829-41a9-bab8-a07012fd2880","year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.298139Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:4593a436641d3a30665d295d2aabc6b5c5ea81f94fee1effa71247c7f2a80876","observation_id":"96c1bc5c-be2d-424d-b62f-1c8a15e68482","resolution":{"observed_at":"2026-08-07T14:52:32.634095Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.369006Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.369006Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:283659a34e833b96ecd92f6abcf2a18ff00eff71612d43ce5154fd482d24fa89","observation_id":"daf4a831-a6f4-4279-b27f-ce3504892117","resolution":{"observed_at":"2026-08-07T14:52:29.369006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.442168Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.442168Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:dec9a099be2141e78e46e4c25d48bf27b6f76e33be835a5cd945921a36785624","observation_id":"af3517cd-6827-4fff-a2d1-0ce3ba1f690d","resolution":{"observed_at":"2026-08-07T14:52:29.442168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14962","last_updated":"2024-12-23T19:01:23Z","snapshot_observed_at":"2026-08-07T04:45:35.203824Z","submitted_at":"2024-11-22T14:21:18Z","title":"LLM for Barcodes: Generating Diverse Synthetic Data for Identity Documents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14962","snapshot_observed_at":"2026-08-07T14:52:29.517727Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.517727Z"},"links":{"cited_paper":"/paper/2411.14962","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:10db1b456455e13dae08e7f780a363ec1354f8cce51ec22697af589ebc6fb33d","observation_id":"b52a48f3-7d90-42a4-a53f-80ae4217600b","resolution":{"observed_at":"2026-08-07T14:52:29.517727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.573151Z","title":"Review of reference generation methods in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.573151Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:d0c4c8e572e5a84aa6a209cbf010ac3aaa1e9d7711b2e5568f42833bab6d1759","observation_id":"fa05beb8-01b6-4ffc-96c4-a948895bb029","resolution":{"observed_at":"2026-08-07T14:52:29.573151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:29.624489Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.624489Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:407736c0870c4d5e979b74abf133d9ea321200adfb5fa31005fc462b8c3baa4e","observation_id":"7090315f-e05e-42f2-8158-447c8d1cf6fd","resolution":{"observed_at":"2026-08-07T14:52:29.624489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16977","last_updated":"2025-04-23T17:28:38Z","snapshot_observed_at":"2026-08-07T15:59:47.335756Z","submitted_at":"2025-04-23T17:28:38Z","title":"Tokenization Matters: Improving Zero-Shot NER for Indic Languages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16977","snapshot_observed_at":"2026-08-07T14:52:29.703994Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.703994Z"},"links":{"cited_paper":"/paper/2504.16977","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:2f0b76841c01d5edee9612b625db8d75f042658c3dceeee0f51f4726b5336bed","observation_id":"5b2fa618-4552-4e31-8336-e1f9e5468b8e","resolution":{"observed_at":"2026-08-07T14:52:29.703994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13108","last_updated":"2025-04-23T17:13:28Z","snapshot_observed_at":"2026-08-07T18:07:36.239752Z","submitted_at":"2025-02-18T18:20:37Z","title":"Clinical QA 2.0: Multi-Task Learning for Answer Extraction and Categorization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13108","snapshot_observed_at":"2026-08-07T14:52:29.883990Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.883990Z"},"links":{"cited_paper":"/paper/2502.13108","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:193ff5cbbf6ec4604f6a01289eba3732c9f2c5fc23ca974790f2d995577a2ded","observation_id":"686ffdf8-d601-414b-ac05-0c6020e20bd6","resolution":{"observed_at":"2026-08-07T14:52:29.883990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.17759","last_updated":"2024-12-23T18:15:19Z","snapshot_observed_at":"2026-07-06T20:12:18.144353Z","submitted_at":"2024-12-23T18:15:19Z","title":"Survey of Large Multimodal Model Datasets, Application Categories and Taxonomy","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.17759","snapshot_observed_at":"2026-08-07T14:52:29.947591Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:29.947591Z"},"links":{"cited_paper":"/paper/2412.17759","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:3e608a145db4d17cf9c99424facfb99642849642165517ab1dd8c054f3676966","observation_id":"c9cdb51a-fff7-4bf0-8352-94db76df3dcc","resolution":{"observed_at":"2026-08-07T14:52:29.947591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:30.022676Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.022676Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:3a5efa26c6f121d617b99ea5a66b9acd5f963e44f275a7cb30bf29d4054a40fb","observation_id":"69935a2f-d505-4a50-b4ac-003d4e18c356","resolution":{"observed_at":"2026-08-07T14:52:30.022676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01263","last_updated":"2024-04-01T11:50:35Z","snapshot_observed_at":"2026-08-03T00:58:55.865010Z","submitted_at":"2023-08-02T16:30:40Z","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01263","snapshot_observed_at":"2026-08-07T14:52:30.095909Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.095909Z"},"links":{"cited_paper":"/paper/2308.01263","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:ffa01218772b61aa81453b13d5cfe77c61c40a7ff444ac674f5c9bcf06699201","observation_id":"52b6447c-c493-4b19-89bc-6796518e7df5","resolution":{"observed_at":"2026-08-07T14:52:30.095909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:30.159695Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.159695Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:c9d414dc6845791e50cd30d69cbab4ba2495193d5dce5f3e4403aa6bb74c3532","observation_id":"5058c140-2e80-4716-b643-cb7382fe309d","resolution":{"observed_at":"2026-08-07T14:52:30.159695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13136","last_updated":"2024-01-23T23:12:09Z","snapshot_observed_at":"2026-07-06T17:19:39.669611Z","submitted_at":"2024-01-23T23:12:09Z","title":"The Language Barrier: Dissecting Safety Challenges of LLMs in Multilingual Contexts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13136","snapshot_observed_at":"2026-08-07T14:52:30.239795Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.239795Z"},"links":{"cited_paper":"/paper/2401.13136","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:98dac9568321d4455283afc263faf9ce1650435c293cf4c97b3b2df702a00b5c","observation_id":"2b1e54c6-b92e-48a4-aefe-4140254a9bb5","resolution":{"observed_at":"2026-08-07T14:52:30.239795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.392450Z","title":null,"venue":null,"work_id":"d453b52c-cb07-4882-8de2-78e86ada9bee","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.305517Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:10f191189da780f51012d42fceca71cd0f5589218106019470e67fb2929ade32","observation_id":"4a3b8bd0-cde7-4ee9-a71d-07972626824b","resolution":{"observed_at":"2026-08-07T14:52:32.452987Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.08377","last_updated":"2023-10-09T15:52:30Z","snapshot_observed_at":"2026-08-04T12:11:05.713517Z","submitted_at":"2023-05-15T06:24:45Z","title":"Text Classification via Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.08377","snapshot_observed_at":"2026-08-07T14:52:30.375712Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.375712Z"},"links":{"cited_paper":"/paper/2305.08377","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:349135bfbc6ff8ce5af344736608c5cefde2ea4d5ae897c7c55b28ef7b935d3c","observation_id":"6875d0ee-2f5a-41e4-a4c4-65e207abcc1f","resolution":{"observed_at":"2026-08-07T14:52:30.375712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:30.469185Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.469185Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:3de59c07609a0f864c7a97db1456caf1b1f38b70498a0da47c834f4051429b41","observation_id":"7e4a57dd-0d8c-41dd-b295-f35e2e2bdbe4","resolution":{"observed_at":"2026-08-07T14:52:30.469185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08676","last_updated":"2024-06-24T08:50:22Z","snapshot_observed_at":"2026-07-06T17:59:29.408432Z","submitted_at":"2024-04-06T15:01:47Z","title":"ALERT: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08676","snapshot_observed_at":"2026-08-07T14:52:30.530713Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.530713Z"},"links":{"cited_paper":"/paper/2404.08676","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:0b5c8631ae9e624d0b69e75859c060027e377a2153cebeb9140647d4fa5e3583","observation_id":"9ad2ab34-9314-4d9f-966e-fba96dd00264","resolution":{"observed_at":"2026-08-07T14:52:30.530713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.221140Z","title":null,"venue":null,"work_id":"50eb00b4-ca11-4c0b-b9f9-fa3f31e7f0d1","year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.572397Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:7691ae4b2e60c7e77fc4e14d91c48a2a12dba5a25f44321f013849c173d96c44","observation_id":"fff2e369-c818-4d8d-b4a1-3064e957e2bb","resolution":{"observed_at":"2026-08-07T14:52:32.284403Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T14:52:30.616707Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.616707Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:c153d9a7560129414debdd0b829be97b6fcfd923f5925b42523a427c74c5a5a0","observation_id":"8f548a86-a549-49d8-bf54-2da948643e91","resolution":{"observed_at":"2026-08-07T14:52:30.616707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00905","last_updated":"2024-06-20T14:15:23Z","snapshot_observed_at":"2026-08-06T23:53:34.059119Z","submitted_at":"2023-10-02T05:23:34Z","title":"All Languages Matter: On the Multilingual Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00905","snapshot_observed_at":"2026-08-07T14:52:30.662928Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.662928Z"},"links":{"cited_paper":"/paper/2310.00905","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5d82bff5fa9c554b6ded01efb797b4cf24628e736a76ce2befaf3b66a5b655ee","observation_id":"5661c39f-23e3-44c4-8c81-356c52133435","resolution":{"observed_at":"2026-08-07T14:52:30.662928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13387","last_updated":"2023-09-04T01:47:30Z","snapshot_observed_at":"2026-07-06T16:10:23.328130Z","submitted_at":"2023-08-25T14:02:12Z","title":"Do-Not-Answer: A Dataset for Evaluating Safeguards in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.13387","snapshot_observed_at":"2026-08-07T14:52:30.727155Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.727155Z"},"links":{"cited_paper":"/paper/2308.13387","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:dc2cb13e2ad7df66001b44513697f08cc28664a57b6a7299e0f4516b786924f9","observation_id":"8dea832e-a657-4a75-9287-1323bcb16dd5","resolution":{"observed_at":"2026-08-07T14:52:30.727155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.10523","last_updated":"2024-12-07T09:33:20Z","snapshot_observed_at":"2026-08-06T20:13:51.493067Z","submitted_at":"2024-05-17T04:05:05Z","title":"Adaptable and Reliable Text Classification using Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.10523","snapshot_observed_at":"2026-08-07T14:52:30.798154Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.798154Z"},"links":{"cited_paper":"/paper/2405.10523","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:9f1bf2379a8c6be5839f634a5b4d4f53d0625347f0474112a5ce4b350be393b0","observation_id":"a4d6c8c2-935d-472d-a3e9-0023f6223a01","resolution":{"observed_at":"2026-08-07T14:52:30.798154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:32.071045Z","title":null,"venue":null,"work_id":"7689dfaa-e966-4cce-869b-006d853f2565","year":2025},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.869862Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:a5844e08ba6d8adc6bf87257b4ceaea9d5a1fb6028bdeb38fdd8e5dd3b8d4ee8","observation_id":"c99df6db-c545-4525-8675-69f32fd72396","resolution":{"observed_at":"2026-08-07T14:52:32.143629Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.950366Z","title":null,"venue":null,"work_id":"336d2109-484f-4099-8de2-13ab95e24d4d","year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.926313Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:0eb8aa46dd9cd5c6fcf60e2a00b4d2729d6a305748dbe94cc2609209343082f1","observation_id":"fa468d64-aa82-41e4-93b2-5c3567b007d3","resolution":{"observed_at":"2026-08-07T14:52:31.979578Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14598","last_updated":"2025-03-01T21:45:36Z","snapshot_observed_at":"2026-07-06T18:34:29.513732Z","submitted_at":"2024-06-20T17:56:07Z","title":"SORRY-Bench: Systematically Evaluating Large Language Model Safety Refusal","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14598","snapshot_observed_at":"2026-08-07T14:52:30.965756Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:30.965756Z"},"links":{"cited_paper":"/paper/2406.14598","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:5c28396ce1db3c2c0ef34852daf87c50d28a73d09a0b2dc739bcaf68b689f058","observation_id":"6d811496-4a3b-4978-864f-92c658110c44","resolution":{"observed_at":"2026-08-07T14:52:30.965756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.008092Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.008092Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:02bd08287c52d250c973b5a6d40a601f14cb01f44ec2fe3f2ea12454742b2a8e","observation_id":"4b32d5de-e91b-44ca-bec3-3bf79eb58ad6","resolution":{"observed_at":"2026-08-07T14:52:31.008092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.085302Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.085302Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:9733bcc099efe968689988c407c4ca6c91d19babda8182fac494d79fe96fcb9a","observation_id":"6df9c3e1-659f-4b29-9f90-bbf5a8fe15e0","resolution":{"observed_at":"2026-08-07T14:52:31.085302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.07045","last_updated":"2024-06-24T04:04:21Z","snapshot_observed_at":"2026-08-06T09:15:44.308151Z","submitted_at":"2023-09-13T15:56:50Z","title":"SafetyBench: Evaluating the Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.07045","snapshot_observed_at":"2026-08-07T14:52:31.148227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.148227Z"},"links":{"cited_paper":"/paper/2309.07045","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:347d0a068073efe5bbbab1911e7f400fac910913f0c9a5a24634d0d109dc3412","observation_id":"a444735e-47da-4991-9cee-16ece478ec49","resolution":{"observed_at":"2026-08-07T14:52:31.148227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:52:31.213008Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.213008Z"},"links":{"citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:406f18d23a761b783ca841ad1160e10fdb165a37276699f5b59e46d7bf11bab3","observation_id":"a1c9744b-1e2d-4831-be5f-977c415f43c2","resolution":{"observed_at":"2026-08-07T14:52:31.213008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-07-06T15:59:23.019044Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-07T14:52:31.262024Z","title":"Zico Kolter, and Matt Fredrikson","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:31.262024Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:b0a75638e336f1b4f1952e58f6abc39e024ada2aae0364361753be8da7f155b4","observation_id":"419e77c1-3e29-460a-b529-1c910287c98e","resolution":{"observed_at":"2026-08-07T14:52:31.262024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":58,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 0 inbound Pith citation observations for arXiv:2505.17332."}