{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F2TSDDAHZY4IOEA3TZ77KQMIDF","short_pith_number":"pith:F2TSDDAH","schema_version":"1.0","canonical_sha256":"2ea7218c07ce3887101b9e7ff54188195c21d003dade77d0e51b82e21508bed9","source":{"kind":"arxiv","id":"2411.01077","version":5},"attestation_state":"computed","paper":{"title":"Emoji Attack: Enhancing Jailbreak Attacks Against Judge LLM Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"N. Benjamin Erichson, Yuqi Liu, Zhipeng Wei","submitted_at":"2024-11-01T23:18:32Z","abstract_excerpt":"Jailbreaking techniques trick Large Language Models (LLMs) into producing restricted output, posing a potential threat. One line of defense is to use another LLM as a Judge to evaluate the harmfulness of generated text. However, we reveal that these Judge LLMs are vulnerable to token segmentation bias, an issue that arises when delimiters alter the tokenization process, splitting words into smaller sub-tokens. This alters the embeddings of the entire sequence, reducing detection accuracy and allowing harmful content to be misclassified as safe. In this paper, we introduce Emoji Attack, a novel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01077","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-01T23:18:32Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f260a360b5da63de6e30f8eb538a7acca170919b968b1a1db0d3547a0c0c9505","abstract_canon_sha256":"6ad4ac92199567556279e0baa4b4f44423cf31dd9f72403eee2c1fa3340002c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:54:42.448286Z","signature_b64":"St9EJ/PfH1uCK9rKQ+0P8q/hiP3CqgMpavDS0JPjnK9nFcxqUoW2jbTEJ3IFLfJFD7knX8gS3wpu7ZeifbCRDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ea7218c07ce3887101b9e7ff54188195c21d003dade77d0e51b82e21508bed9","last_reissued_at":"2026-07-05T11:54:42.447800Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:54:42.447800Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Emoji Attack: Enhancing Jailbreak Attacks Against Judge LLM Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"N. Benjamin Erichson, Yuqi Liu, Zhipeng Wei","submitted_at":"2024-11-01T23:18:32Z","abstract_excerpt":"Jailbreaking techniques trick Large Language Models (LLMs) into producing restricted output, posing a potential threat. One line of defense is to use another LLM as a Judge to evaluate the harmfulness of generated text. However, we reveal that these Judge LLMs are vulnerable to token segmentation bias, an issue that arises when delimiters alter the tokenization process, splitting words into smaller sub-tokens. This alters the embeddings of the entire sequence, reducing detection accuracy and allowing harmful content to be misclassified as safe. In this paper, we introduce Emoji Attack, a novel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01077","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01077/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01077","created_at":"2026-07-05T11:54:42.447869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01077v5","created_at":"2026-07-05T11:54:42.447869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01077","created_at":"2026-07-05T11:54:42.447869+00:00"},{"alias_kind":"pith_short_12","alias_value":"F2TSDDAHZY4I","created_at":"2026-07-05T11:54:42.447869+00:00"},{"alias_kind":"pith_short_16","alias_value":"F2TSDDAHZY4IOEA3","created_at":"2026-07-05T11:54:42.447869+00:00"},{"alias_kind":"pith_short_8","alias_value":"F2TSDDAH","created_at":"2026-07-05T11:54:42.447869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2505.10846","citing_title":"AutoRAN: Automated Hijacking of Safety Reasoning in Large Reasoning Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01473","citing_title":"SelfGrader: LLM Jailbreak Detection via Anchored Token-Level Logits","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12548","citing_title":"DeepSeek Robustness Against Semantic-Character Dual-Space Mutated Prompt Injection","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF","json":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF.json","graph_json":"https://pith.science/api/pith-number/F2TSDDAHZY4IOEA3TZ77KQMIDF/graph.json","events_json":"https://pith.science/api/pith-number/F2TSDDAHZY4IOEA3TZ77KQMIDF/events.json","paper":"https://pith.science/paper/F2TSDDAH"},"agent_actions":{"view_html":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF","download_json":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF.json","view_paper":"https://pith.science/paper/F2TSDDAH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01077&json=true","fetch_graph":"https://pith.science/api/pith-number/F2TSDDAHZY4IOEA3TZ77KQMIDF/graph.json","fetch_events":"https://pith.science/api/pith-number/F2TSDDAHZY4IOEA3TZ77KQMIDF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF/action/storage_attestation","attest_author":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF/action/author_attestation","sign_citation":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF/action/citation_signature","submit_replication":"https://pith.science/pith/F2TSDDAHZY4IOEA3TZ77KQMIDF/action/replication_record"}},"created_at":"2026-07-05T11:54:42.447869+00:00","updated_at":"2026-07-05T11:54:42.447869+00:00"}