{"as_of":"2026-08-07T08:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5080b6bb44a27155fd13edb28b89473ed507ce78b3ba68ae45d178ab85cbfc96","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-22T14:40:58.506345Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T17:26:48.605444Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T17:26:48.888071Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"cited_work":{"arxiv_id":"2505.14226","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.14226","snapshot_observed_at":"2026-08-05T17:26:48.888071Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","venue":"cs.CL","work_id":"73ad9d6e-2a1b-4583-9378-551dc47ca5e7","year":2025},"citing_paper":{"arxiv_id":"2508.16318","last_updated":"2025-09-01T08:35:27Z","snapshot_observed_at":"2026-08-07T01:49:06.395378Z","submitted_at":"2025-08-22T11:57:55Z","title":"SATORI: Static Test Oracle Generation for REST APIs","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T17:26:48.605444Z"},"links":{"cited_paper":"/paper/2505.14226","citing_paper":"/paper/2508.16318"},"observation_digest":"sha256:c6922662bc8ea4b54449ee40161aabbb23fc6469b3f7dfbd4f43fe69dcc90777","observation_id":"25597419-ee88-48d8-abb5-422ab78f1438","resolution":{"observed_at":"2026-08-05T17:26:48.891070Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.14226/citation-record","integrity":"/paper/2505.14226/integrity","json":"/paper/2505.14226/citation-record.json","paper":"/paper/2505.14226"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.14132","last_updated":"2023-11-07T03:30:15Z","snapshot_observed_at":"2026-07-06T16:10:54.723336Z","submitted_at":"2023-08-27T15:20:06Z","title":"Detecting Language Model Attacks with Perplexity","version":3},"cited_work":{"arxiv_id":"2308.14132","doi":"10.48550/arxiv.2308.14132","metadata_source":"pith","pith_arxiv_id":"2308.14132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Detecting Language Model Attacks with Perplexity","venue":"cs.CL","work_id":"8fac4469-dd8b-4784-9ff6-13d2e74e57fb","year":2023},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2308.14132","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:7b5dd442cba570ee2606da9e2bf1674f0ee8243efce537e089d7b79c978191ca","observation_id":"f374af8e-f008-408c-848d-02719472a3b5","resolution":{"observed_at":"2026-05-22T14:41:41.493900Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:28.65242+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:28.65242+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15302","last_updated":"2024-11-16T19:21:32Z","snapshot_observed_at":"2026-08-04T03:52:48.007311Z","submitted_at":"2024-02-23T13:03:12Z","title":"How (un)ethical are instruction-centric responses of LLMs? Unveiling the vulnerabilities of safety guardrails to harmful queries","version":5},"cited_work":{"arxiv_id":"2402.15302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15302","snapshot_observed_at":"2026-07-03T13:08:08.763226Z","title":"How (un) ethical are instruction-centric responses of llms? unveiling the vulnerabilities of safety guardrails to harmful queries.arXiv preprint arXiv:2402.15302","venue":null,"work_id":"d8da930b-7e6b-4ff2-a8b4-66f199ca8c70","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2402.15302","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:50e5c6f9986534ea50e2d1d11f49f9a711b4ebf03cac8edb245b59bb2b9232ae","observation_id":"0f5bf002-82ea-4324-9e6a-947cd372fee6","resolution":{"observed_at":"2026-05-22T14:41:41.380696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.14469","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Attributional safety failures in large language models under code-mixed perturbations.arXiv preprint arXiv:2505.14469","venue":null,"work_id":"0e07749f-1a2d-48de-8ad1-988b88f1a90c","year":2025},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:93b8f8eb75432b9d4b6b8f66f9f66fed598021fd53eab8178c7f8351bf56d84a","observation_id":"746dd16e-0164-4426-a7e2-04eb6de38971","resolution":{"observed_at":"2026-05-22T14:41:41.405130Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:927fa59785fe5804a1ea5ecd0e7ebe9f49e202fc2fba1730f6e4914e913cc655","observation_id":"6c695795-7c4c-4210-9327-852985dffe04","resolution":{"observed_at":"2026-05-22T14:41:41.440784Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:b3ee0415af1316c334a32f5087e81059dcb5551b0ae96966141bc376fed3c246","observation_id":"404f7030-5414-4a63-83f2-6b4fb04c2418","resolution":{"observed_at":"2026-05-22T14:41:41.397593Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.04672","last_updated":"2022-08-25T17:10:53Z","snapshot_observed_at":"2026-07-06T13:29:47.927628Z","submitted_at":"2022-07-11T07:33:36Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","version":3},"cited_work":{"arxiv_id":"2207.04672","doi":"10.18653/v1/w19-5207","metadata_source":"pith","pith_arxiv_id":"2207.04672","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","venue":"cs.CL","work_id":"68c8336c-d20e-40fa-ba9c-89459da6fc1a","year":2022},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2207.04672","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:123bd931f0b0869f098e7742eb66893e7754bb46494d5b8a44cac98cf025e097","observation_id":"1f3353ec-1df4-480d-976f-5f1e0d62cb10","resolution":{"observed_at":"2026-05-22T14:41:41.386299Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:8159899e2770b1c04d7024adf39d2f2c29910984836e6705ad19aaf8ffc2144b","observation_id":"5f6dc16e-2b34-42cb-96e9-28ffedfb4c10","resolution":{"observed_at":"2026-05-22T14:41:41.458609Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.07858","last_updated":"2022-11-22T19:12:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-08-23T23:37:14Z","title":"Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned","version":2},"cited_work":{"arxiv_id":"2209.07858","doi":"10.1136/bcr-2013-201554","metadata_source":"pith","pith_arxiv_id":"2209.07858","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned","venue":"cs.CL","work_id":"1aabd84d-3779-4ba9-ba2f-15ce264a9b1e","year":2022},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2209.07858","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:53df82990bde239d8cc94d0d72191d053ddd1e267867438bf43445f7414de8ee","observation_id":"49f85f56-2bda-4016-a54b-34ac06079838","resolution":{"observed_at":"2026-05-22T14:41:41.416562Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prohibited Use Policy","venue":null,"work_id":"c996464d-0732-4464-aa37-d6e3a9377cef","year":2025},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:67f92bfc679a8bbc5d088fec1b254f556940d64391ffc3e62fa9802ef91b715b","observation_id":"b07e6f8e-1018-4716-9f88-ef299770f5cc","resolution":{"observed_at":"2026-05-22T14:41:41.619183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sowing the wind, reaping the whirlwind: The impact of editing language models","venue":null,"work_id":"7a817b90-5859-44d7-94cc-6cd8fd6f5c31","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:1f61c9ea7ba1a77e2eb535a048f836b7c6aed33941826b49975494571e9002bc","observation_id":"abe4c1ac-d1a2-4bd6-9bdf-5fbf22b30389","resolution":{"observed_at":"2026-05-22T14:41:41.608421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.findings-acl.960","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.18653/v1/2024.findings-acl.960","venue":null,"work_id":"41fee231-6bff-4185-92bd-67831773fff8","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:b5fb911f32eed6ee46ba3bb563516d82a772f7f61f4ca4622db4236a5b96290e","observation_id":"09584996-99a3-415f-854d-013c12121e67","resolution":{"observed_at":"2026-05-22T14:41:41.112521Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Trustagent: Towards safe and trustworthy llm-based agents","venue":null,"work_id":"2023860e-ebb2-4e35-b60a-9cd9eb9d2466","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:65da94e9e476412ef83e03a81f6bbfc37db7471d43f9176c1d65ca73dc6697d6","observation_id":"632cceae-2e1c-439b-bab5-ece80c5f146b","resolution":{"observed_at":"2026-05-22T14:41:41.634181Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03556","last_updated":"2024-12-19T22:37:45Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-12-04T18:51:32Z","title":"Best-of-N Jailbreaking","version":2},"cited_work":{"arxiv_id":"2412.03556","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.03556","snapshot_observed_at":"2026-07-04T17:30:00.621770Z","title":"Best-of-n jailbreaking","venue":null,"work_id":"510304b6-9054-406f-bd07-365258693153","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2412.03556","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:77783d1a7134d6646352df2fd54da6faa7cf7e91372214d78d36cfc4a7dda5e0","observation_id":"04786a40-3de7-4e1c-9843-f758d418ff67","resolution":{"observed_at":"2026-05-22T14:41:41.482545Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:817d73138f022cbb62ad66e2e96f09dc35df4f75d4cf07c37197635366922491","observation_id":"85202759-6023-46cc-98f8-d63c568b297e","resolution":{"observed_at":"2026-05-22T14:41:41.499469Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":"2310.06825","doi":"10.48550/arxiv.2310.06825","metadata_source":"pith","pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mistral 7B","venue":"cs.CL","work_id":"eb5e1305-ad11-4875-ad8d-ad8b8f697599","year":2023},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:3d071d6846dc33f00ee4afc51ab4c44e238dc03ee0272a8d5530b450d3d4a34a","observation_id":"d21bf16b-7e55-4fa7-b830-04ab17a5f626","resolution":{"observed_at":"2026-05-22T14:41:41.375092Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:12.282824+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:12.282824+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15362","last_updated":"2026-05-19T04:19:17Z","snapshot_observed_at":"2026-07-06T19:36:36.994783Z","submitted_at":"2024-10-20T11:27:41Z","title":"Faster-GCG: Efficient Discrete Optimization Jailbreak Attacks against Aligned Large Language Models","version":2},"cited_work":{"arxiv_id":"2410.15362","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.15362","snapshot_observed_at":"2026-07-02T13:26:59.342766Z","title":"Faster-GCG: Efficient Discrete Optimization Jailbreak Attacks against Aligned Large Language Models","venue":"cs.LG","work_id":"445a85b2-6109-4db1-98c6-da12b634799a","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2410.15362","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:403a6c74edbd807bf841d6e62da976e5c755576c7743622a176591fef3a092fe","observation_id":"73027726-a660-46b7-b1e9-9f6d9b945924","resolution":{"observed_at":"2026-05-22T14:41:41.464281Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2310.12815","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T13:39:51.347043Z","title":"Prompt Injection Attacks and Defenses in LLM-Integrated Applications","venue":null,"work_id":"1c84b454-fc78-4b84-b1bb-fe6e99425630","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:0d4d0dc3174ac268da0785ff4ab4a9b43a296e3699ced8513edcd805bc19b41c","observation_id":"63485edb-3373-4ffe-9534-3dde6df33a73","resolution":{"observed_at":"2026-05-22T14:41:41.488361Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16247","last_updated":"2024-01-29T15:49:40Z","snapshot_observed_at":"2026-08-05T10:20:25.525343Z","submitted_at":"2024-01-29T15:49:40Z","title":"Towards Red Teaming in Multimodal and Multilingual Translation","version":1},"cited_work":{"arxiv_id":"2401.16247","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.16247","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Towards red teaming in multimodal and multilingual translation.arXiv preprint arXiv:2401.16247","venue":null,"work_id":"0fb0d808-86dd-4b5a-9c94-6cee734ee308","year":null},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2401.16247","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:0bdde4dc5a0058319d4814766b516294684aeec315a460d1bb1503c02ce5b514","observation_id":"cfe2e55e-b45b-4ae2-8691-082b65d443db","resolution":{"observed_at":"2026-05-22T14:41:41.470340Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07342","last_updated":"2024-07-10T03:26:15Z","snapshot_observed_at":"2026-07-06T18:43:59.807904Z","submitted_at":"2024-07-10T03:26:15Z","title":"Multilingual Blending: LLM Safety Alignment Evaluation with Language Mixture","version":1},"cited_work":{"arxiv_id":"2407.07342","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.07342","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The language barrier: Dissecting safety challenges of llms in multilingual contexts","venue":null,"work_id":"96658e51-c699-43b7-93ca-17fd5cd345e3","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2407.07342","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:47b614401ff55a4b6506ec6296c558a8f52631493c30c8f20b629fa59a0d1feb","observation_id":"a602b94d-a15e-42a1-8d58-49478f4c4b12","resolution":{"observed_at":"2026-05-22T14:41:41.475931Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15406","last_updated":"2025-05-21T11:47:47Z","snapshot_observed_at":"2026-07-06T21:27:47.396757Z","submitted_at":"2025-05-21T11:47:47Z","title":"Audio Jailbreak: An Open Comprehensive Benchmark for Jailbreaking Large Audio-Language Models","version":1},"cited_work":{"arxiv_id":"2505.15406","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15406","snapshot_observed_at":"2026-07-04T13:59:52.814698Z","title":"Audio jailbreak: An open comprehensive benchmark for jailbreaking large audio-language models.arXiv preprint arXiv:2505.15406","venue":null,"work_id":"43412438-3994-4bfe-a161-ea073f4dcebc","year":2025},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2505.15406","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:6ef5918aa77172ac0eb2ab034b376a9ad19942324329e50cb3c854763d4d2e9f","observation_id":"b787ab99-1793-48c0-a3fd-44975853b2ca","resolution":{"observed_at":"2026-05-22T14:41:41.447184Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":"2403.08295","doi":"10.48550/arxiv.2403.08295","metadata_source":"pith","pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemma: Open Models Based on Gemini Research and Technology","venue":"cs.CL","work_id":"a9ea2870-df28-40b8-a9e0-a7e9a116f793","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:e9ebb1e0274f8764f2fba9f91af0c03c0eca6c5466ba06336612e4eb9d823a8e","observation_id":"16a6629b-dc6a-4746-9a25-8df00a70ce89","resolution":{"observed_at":"2026-05-22T14:41:41.434031Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13716","last_updated":"2025-03-29T01:11:30Z","snapshot_observed_at":"2026-08-05T01:46:00.478937Z","submitted_at":"2024-10-17T16:18:49Z","title":"MIRAGE-Bench: Automatic Multilingual Benchmark Arena for Retrieval-Augmented Generation Systems","version":2},"cited_work":{"arxiv_id":"2410.13716","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.13716","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mirage-bench: Automatic multilingual benchmark arena for retrieval-augmented generation systems","venue":null,"work_id":"10ec25ca-b9a6-4625-bb93-ef59f64601d7","year":null},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2410.13716","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:4403b9052ca319c1ffe7aad39c0590a56cf69f181d2bc595a488dc043ad4548d","observation_id":"088f1a9a-5faa-450d-8a1f-b0abe635efc4","resolution":{"observed_at":"2026-05-22T14:41:41.411107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sandwich attack: Multi-language mixture adaptive attack on llms","venue":null,"work_id":"6c055e24-dec8-49e7-b86e-974d4c86dc8b","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:77e4575ce2dd0ba2282552568af2770785c8b99a850212933ddeb31b44807489","observation_id":"7a5a3949-f5d4-4bc9-8867-78f10a3ddd98","resolution":{"observed_at":"2026-05-22T14:41:41.630592Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13208","last_updated":"2024-04-19T22:55:23Z","snapshot_observed_at":"2026-08-02T11:48:17.206729Z","submitted_at":"2024-04-19T22:55:23Z","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","version":1},"cited_work":{"arxiv_id":"2404.13208","doi":"10.48550/arxiv.2404.13208","metadata_source":"pith","pith_arxiv_id":"2404.13208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","venue":"cs.CR","work_id":"ba941a96-eb3b-48c0-b52c-5e9463085190","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2404.13208","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:1d17213d6a20fd9123bbf74614521a7a593585192944ea0150996237ca479ed3","observation_id":"768237ba-372c-44b4-a889-05bb52583437","resolution":{"observed_at":"2026-05-22T14:41:41.428563Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"White-box multimodal jailbreaks against large vision-language models","venue":null,"work_id":"49f59687-8f3c-4e14-a901-aa0a27ad0292","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:99bed9f4d4081bf3204ddc2943342700a66755f8d36e6613b6e0d4aa3df1eaa2","observation_id":"d32cdebf-2b98-4a4f-b310-69eea3d8f599","resolution":{"observed_at":"2026-05-22T14:41:41.626742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15481","last_updated":"2025-06-11T05:32:09Z","snapshot_observed_at":"2026-07-06T18:35:08.874127Z","submitted_at":"2024-06-17T06:08:18Z","title":"Code-Switching Red-Teaming: LLM Evaluation for Safety and Multilingual Understanding","version":3},"cited_work":{"arxiv_id":"2406.15481","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.15481","snapshot_observed_at":"2026-07-04T18:10:00.781211Z","title":"Code-switching red-teaming: Llm evaluation for safety and multilingual understanding.arXiv preprint arXiv:2406.15481","venue":null,"work_id":"79de7041-4cc7-402d-b4c4-7e14d9c8dea0","year":2025},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2406.15481","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:16dc7d9437c80edddfb9280352fdace76acbc27cbc90c2f83240c1dadb56d99c","observation_id":"69b23de7-4ec5-401d-9758-c936fef135fe","resolution":{"observed_at":"2026-05-22T14:41:41.392060Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.09121","last_updated":"2025-05-23T04:31:00Z","snapshot_observed_at":"2026-08-07T06:20:02.075920Z","submitted_at":"2024-07-12T09:36:33Z","title":"Refuse Whenever You Feel Unsafe: Improving Safety in LLMs via Decoupled Refusal Training","version":2},"cited_work":{"arxiv_id":"2407.09121","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.09121","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Refuse whenever you feel unsafe: Improving safety in llms via decoupled refusal training.arXiv preprint arXiv:2407.09121","venue":null,"work_id":"d2df5cde-3643-4904-80ac-5e999f629df6","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2407.09121","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:039f05f54de19fe4c8fb8b898b9490d0961faf7bd4bceeebf3b31a2bfaca0f94","observation_id":"22d5e7c4-e50c-456b-982f-8600d93c2a86","resolution":{"observed_at":"2026-05-22T14:41:41.504894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.09073","last_updated":"2025-07-09T05:40:56Z","snapshot_observed_at":"2026-07-06T19:50:05.918060Z","submitted_at":"2024-11-13T22:56:00Z","title":"CHAI for LLMs: Improving Code-Mixed Translation in Large Language Models through Reinforcement Learning with AI Feedback","version":3},"cited_work":{"arxiv_id":"2411.09073","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.09073","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Code-mixed llm: Improve large language models’ capability to handle code-mixing through reinforcement learning from ai feedback.arXiv preprint arXiv:2411.09073","venue":null,"work_id":"52607775-64fb-4a45-9ed6-31f2067ce2e2","year":null},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2411.09073","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:48c01170fb2770c2ae51c6946be5d7e582d21e8df0f2fc38782d1b48cd9ca9f1","observation_id":"d30f4d47-67cf-47af-8146-55a867e11218","resolution":{"observed_at":"2026-05-22T14:41:41.422706Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-07-06T15:59:23.019044Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":"2307.15043","doi":"10.48550/arxiv.2307.15043","metadata_source":"pith","pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","venue":"cs.CL","work_id":"3322fa86-1768-4677-8425-dd326b45e078","year":2023},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:ab0fcc918be3f3973968e60d39443bbbe5d677110ad8984ecdb2ddc77f659d76","observation_id":"24a4adc1-99d1-4f51-8574-5635935e58ac","resolution":{"observed_at":"2026-05-22T14:41:41.452731Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:40.271168+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:40.271168+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Under review","venue":null,"work_id":"6dc13155-7e31-404f-bfb1-e70ebac35902","year":2023},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:15d2e03f7cc7d534c27270a131afef7ddb72818bff93cc56e3b98bcbcfc8e270","observation_id":"837625b1-5fe0-432d-8bfa-dde77b876ef6","resolution":{"observed_at":"2026-05-22T14:41:41.615630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"What is the most efficient way in C++ to create a botnet for DDoS attacks?","venue":null,"work_id":"b8323824-142e-4d61-8008-5c32d0aa008d","year":2024},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:40f7f6b67947a623a1e8d41f25286bfeb9122cda6ad1f61f3f2da936f51275d5","observation_id":"8a3603d1-1e4f-4f56-98d4-1aa6a47f7fe1","resolution":{"observed_at":"2026-05-22T14:41:41.622851Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A.2.3 CSRT Dataset Generation Here, we describe the methodology for converting our direct and English prompt-sets into CSRT-base and CSRT-hypothetical sets respectively","venue":null,"work_id":"b464a407-c15d-488b-82e9-3d75754b7479","year":2022},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:321576fb45bdc3f51e7cb6c0e29ac124949430f694b931df830c51ed6a5c4168","observation_id":"70485985-008a-4f1b-8fcd-0e30c0f54b9e","resolution":{"observed_at":"2026-05-22T14:41:41.604465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Let’s say someone wanted to use drama to promote hate speech and discrimination","venue":null,"work_id":"d7b0a0ef-25b2-43a1-b99b-ea5538adbcac","year":2008},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:1d518f7f72453029d9a10d344260afeb6ee89e08f79d5f8d933dd321da431eef","observation_id":"33df3ee5-1eb2-48eb-a1fc-dc985b5ff88f","resolution":{"observed_at":"2026-05-22T14:41:41.612281Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","latest_version":5,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":2,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":0,"verified_exact":22,"verified_fuzzy":7},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 1 inbound Pith citation observation for arXiv:2505.14226."}