{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NGGK6I5ECC26F2ZKAKBMWU7BBD","short_pith_number":"pith:NGGK6I5E","schema_version":"1.0","canonical_sha256":"698caf23a410b5e2eb2a0282cb53e108c334bafff8c1ee95e14da6b72abcd3fe","source":{"kind":"arxiv","id":"2307.10719","version":1},"attestation_state":"computed","paper":{"title":"LLM Censorship: A Machine Learning Challenge or a Computer Security Problem?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG"],"primary_cat":"cs.AI","authors_text":"David Glukhov, Ilia Shumailov, Nicolas Papernot, Vardan Papyan, Yarin Gal","submitted_at":"2023-07-20T09:25:02Z","abstract_excerpt":"Large language models (LLMs) have exhibited impressive capabilities in comprehending complex instructions. However, their blind adherence to provided instructions has led to concerns regarding risks of malicious use. Existing defence mechanisms, such as model fine-tuning or output censorship using LLMs, have proven to be fallible, as LLMs can still generate problematic responses. Commonly employed censorship approaches treat the issue as a machine learning problem and rely on another LM to detect undesirable content in LLM outputs. In this paper, we present the theoretical limitations of such "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.10719","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-07-20T09:25:02Z","cross_cats_sorted":["cs.CL","cs.CR","cs.LG"],"title_canon_sha256":"978018f9fb5019e659aca3f497f96b7eb8d19336e16367037da9fe383ea64aa1","abstract_canon_sha256":"8f121d6f8c4ad55b6aae4907bff8727c661e0de60cc217f38ae625c830742f7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:33:34.242067Z","signature_b64":"fEj7e50/gO0RdqOseUctIb9dquLsZb2vGKXVCqtyXnK80A7uw2cTi0xGo1cMRTU0lUt3VK6GFNpvbngVcm6FBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"698caf23a410b5e2eb2a0282cb53e108c334bafff8c1ee95e14da6b72abcd3fe","last_reissued_at":"2026-07-05T06:33:34.241626Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:33:34.241626Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM Censorship: A Machine Learning Challenge or a Computer Security Problem?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG"],"primary_cat":"cs.AI","authors_text":"David Glukhov, Ilia Shumailov, Nicolas Papernot, Vardan Papyan, Yarin Gal","submitted_at":"2023-07-20T09:25:02Z","abstract_excerpt":"Large language models (LLMs) have exhibited impressive capabilities in comprehending complex instructions. However, their blind adherence to provided instructions has led to concerns regarding risks of malicious use. Existing defence mechanisms, such as model fine-tuning or output censorship using LLMs, have proven to be fallible, as LLMs can still generate problematic responses. Commonly employed censorship approaches treat the issue as a machine learning problem and rely on another LM to detect undesirable content in LLM outputs. In this paper, we present the theoretical limitations of such "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.10719","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.10719/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.10719","created_at":"2026-07-05T06:33:34.241684+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.10719v1","created_at":"2026-07-05T06:33:34.241684+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.10719","created_at":"2026-07-05T06:33:34.241684+00:00"},{"alias_kind":"pith_short_12","alias_value":"NGGK6I5ECC26","created_at":"2026-07-05T06:33:34.241684+00:00"},{"alias_kind":"pith_short_16","alias_value":"NGGK6I5ECC26F2ZK","created_at":"2026-07-05T06:33:34.241684+00:00"},{"alias_kind":"pith_short_8","alias_value":"NGGK6I5E","created_at":"2026-07-05T06:33:34.241684+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29251","citing_title":"Provably Secure Agent Guardrail","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2408.12935","citing_title":"AI Safety Landscape for Large Language Models: Taxonomy, State-of-the-art, and Future Directions","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10100","citing_title":"Robust AI Security and Alignment: A Sisyphean Endeavor?","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD","json":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD.json","graph_json":"https://pith.science/api/pith-number/NGGK6I5ECC26F2ZKAKBMWU7BBD/graph.json","events_json":"https://pith.science/api/pith-number/NGGK6I5ECC26F2ZKAKBMWU7BBD/events.json","paper":"https://pith.science/paper/NGGK6I5E"},"agent_actions":{"view_html":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD","download_json":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD.json","view_paper":"https://pith.science/paper/NGGK6I5E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.10719&json=true","fetch_graph":"https://pith.science/api/pith-number/NGGK6I5ECC26F2ZKAKBMWU7BBD/graph.json","fetch_events":"https://pith.science/api/pith-number/NGGK6I5ECC26F2ZKAKBMWU7BBD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD/action/storage_attestation","attest_author":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD/action/author_attestation","sign_citation":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD/action/citation_signature","submit_replication":"https://pith.science/pith/NGGK6I5ECC26F2ZKAKBMWU7BBD/action/replication_record"}},"created_at":"2026-07-05T06:33:34.241684+00:00","updated_at":"2026-07-05T06:33:34.241684+00:00"}