{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SFSOE3CJJXCCGE6GWEEVOOFJLV","short_pith_number":"pith:SFSOE3CJ","schema_version":"1.0","canonical_sha256":"9164e26c494dc42313c6b1095738a95d41b29f35ce54f5f1ef28ad4b28bebfd2","source":{"kind":"arxiv","id":"2409.16727","version":1},"attestation_state":"computed","paper":{"title":"RoleBreak: Character Hallucination as a Jailbreak Attack in Role-Playing Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bo Wang, Dongming Zhao, Jijun Zhang, Jing Liu, Ruifang He, Xu Wang, Yihong Tang, Yuexian Hou","submitted_at":"2024-09-25T08:23:46Z","abstract_excerpt":"Role-playing systems powered by large language models (LLMs) have become increasingly influential in emotional communication applications. However, these systems are susceptible to character hallucinations, where the model deviates from predefined character roles and generates responses that are inconsistent with the intended persona. This paper presents the first systematic analysis of character hallucination from an attack perspective, introducing the RoleBreak framework. Our framework identifies two core mechanisms-query sparsity and role-query conflict-as key factors driving character hall"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.16727","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-25T08:23:46Z","cross_cats_sorted":[],"title_canon_sha256":"8d7badfe0e909ad36cc29f129320405fe0ccd5ebc1d5bad4ff5f6257f28cdd0b","abstract_canon_sha256":"a2d42ac9b3d1849360bae0b23e2f4530e5f8cca2efec64e9f1c6d3eba1f41b39"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:11:38.416242Z","signature_b64":"XNixShZ+HIShWMwnnC4+qoXQaUXauHyHWZvElX4ajbbjKWZaZdYSznfjisWTI9y7LBUBgijXmv9hWlNWmbaJDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9164e26c494dc42313c6b1095738a95d41b29f35ce54f5f1ef28ad4b28bebfd2","last_reissued_at":"2026-07-05T09:11:38.415753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:11:38.415753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RoleBreak: Character Hallucination as a Jailbreak Attack in Role-Playing Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bo Wang, Dongming Zhao, Jijun Zhang, Jing Liu, Ruifang He, Xu Wang, Yihong Tang, Yuexian Hou","submitted_at":"2024-09-25T08:23:46Z","abstract_excerpt":"Role-playing systems powered by large language models (LLMs) have become increasingly influential in emotional communication applications. However, these systems are susceptible to character hallucinations, where the model deviates from predefined character roles and generates responses that are inconsistent with the intended persona. This paper presents the first systematic analysis of character hallucination from an attack perspective, introducing the RoleBreak framework. Our framework identifies two core mechanisms-query sparsity and role-query conflict-as key factors driving character hall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.16727","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.16727/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.16727","created_at":"2026-07-05T09:11:38.415811+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.16727v1","created_at":"2026-07-05T09:11:38.415811+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.16727","created_at":"2026-07-05T09:11:38.415811+00:00"},{"alias_kind":"pith_short_12","alias_value":"SFSOE3CJJXCC","created_at":"2026-07-05T09:11:38.415811+00:00"},{"alias_kind":"pith_short_16","alias_value":"SFSOE3CJJXCCGE6G","created_at":"2026-07-05T09:11:38.415811+00:00"},{"alias_kind":"pith_short_8","alias_value":"SFSOE3CJ","created_at":"2026-07-05T09:11:38.415811+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":94,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV","json":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV.json","graph_json":"https://pith.science/api/pith-number/SFSOE3CJJXCCGE6GWEEVOOFJLV/graph.json","events_json":"https://pith.science/api/pith-number/SFSOE3CJJXCCGE6GWEEVOOFJLV/events.json","paper":"https://pith.science/paper/SFSOE3CJ"},"agent_actions":{"view_html":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV","download_json":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV.json","view_paper":"https://pith.science/paper/SFSOE3CJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.16727&json=true","fetch_graph":"https://pith.science/api/pith-number/SFSOE3CJJXCCGE6GWEEVOOFJLV/graph.json","fetch_events":"https://pith.science/api/pith-number/SFSOE3CJJXCCGE6GWEEVOOFJLV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV/action/storage_attestation","attest_author":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV/action/author_attestation","sign_citation":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV/action/citation_signature","submit_replication":"https://pith.science/pith/SFSOE3CJJXCCGE6GWEEVOOFJLV/action/replication_record"}},"created_at":"2026-07-05T09:11:38.415811+00:00","updated_at":"2026-07-05T09:11:38.415811+00:00"}