{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PSARXI34XQRDTUVWMKTSU6BBN2","short_pith_number":"pith:PSARXI34","schema_version":"1.0","canonical_sha256":"7c811ba37cbc2239d2b662a72a78216e95fb954e5f42fff96ea79b33b72c031f","source":{"kind":"arxiv","id":"2507.07341","version":1},"attestation_state":"computed","paper":{"title":"On the Impossibility of Separating Intelligence from Judgment: The Computational Intractability of Filtering for AI Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.AI","authors_text":"Frauke Kreuter, Greg Gluch, Guy N. Rothblum, Omer Reingold, Sarah Ball, Shafi Goldwasser","submitted_at":"2025-07-09T23:55:35Z","abstract_excerpt":"With the increased deployment of large language models (LLMs), one concern is their potential misuse for generating harmful content. Our work studies the alignment challenge, with a focus on filters to prevent the generation of unsafe information. Two natural points of intervention are the filtering of the input prompt before it reaches the model, and filtering the output after generation. Our main results demonstrate computational challenges in filtering both prompts and outputs. First, we show that there exist LLMs for which there are no efficient prompt filters: adversarial prompts that eli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.07341","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-09T23:55:35Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"d2519ceadcecb3b5b768e97134e524a9418a2c47a48cb0b419ade05c0ef70b4a","abstract_canon_sha256":"d9d9c4bc915675ef8a3f189e63369eb43d35e7c9c1af569d5a1a3fd4a4a6573a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:56.229696Z","signature_b64":"H1N0ErPEvuQkUsB5pq38kicfXIvh6L8yC3TWlmW2/uujvuYCeqFqI59wjBWiieMc43mNSQjeaWv4hfglg0CBDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c811ba37cbc2239d2b662a72a78216e95fb954e5f42fff96ea79b33b72c031f","last_reissued_at":"2026-07-05T11:34:56.229207Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:56.229207Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Impossibility of Separating Intelligence from Judgment: The Computational Intractability of Filtering for AI Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.AI","authors_text":"Frauke Kreuter, Greg Gluch, Guy N. Rothblum, Omer Reingold, Sarah Ball, Shafi Goldwasser","submitted_at":"2025-07-09T23:55:35Z","abstract_excerpt":"With the increased deployment of large language models (LLMs), one concern is their potential misuse for generating harmful content. Our work studies the alignment challenge, with a focus on filters to prevent the generation of unsafe information. Two natural points of intervention are the filtering of the input prompt before it reaches the model, and filtering the output after generation. Our main results demonstrate computational challenges in filtering both prompts and outputs. First, we show that there exist LLMs for which there are no efficient prompt filters: adversarial prompts that eli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.07341","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.07341/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.07341","created_at":"2026-07-05T11:34:56.229266+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.07341v1","created_at":"2026-07-05T11:34:56.229266+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.07341","created_at":"2026-07-05T11:34:56.229266+00:00"},{"alias_kind":"pith_short_12","alias_value":"PSARXI34XQRD","created_at":"2026-07-05T11:34:56.229266+00:00"},{"alias_kind":"pith_short_16","alias_value":"PSARXI34XQRDTUVW","created_at":"2026-07-05T11:34:56.229266+00:00"},{"alias_kind":"pith_short_8","alias_value":"PSARXI34","created_at":"2026-07-05T11:34:56.229266+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08367","citing_title":"Emergence World: A Platform for Evaluating Long-Horizon Multi-Agent Autonomy","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30837","citing_title":"Send a SCOUT First: Pre-hoc Reasoning for Adaptive Detector Allocation in Prompt-Injection Defense","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16776","citing_title":"Distinguishable Deletion: Unifying Knowledge Erasure and Refusal for Large Language Model Unlearning","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23711","citing_title":"Spore: Efficient and Training-Free Privacy Extraction Attack on LLMs via Inference-Time Hybrid Probing","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2","json":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2.json","graph_json":"https://pith.science/api/pith-number/PSARXI34XQRDTUVWMKTSU6BBN2/graph.json","events_json":"https://pith.science/api/pith-number/PSARXI34XQRDTUVWMKTSU6BBN2/events.json","paper":"https://pith.science/paper/PSARXI34"},"agent_actions":{"view_html":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2","download_json":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2.json","view_paper":"https://pith.science/paper/PSARXI34","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.07341&json=true","fetch_graph":"https://pith.science/api/pith-number/PSARXI34XQRDTUVWMKTSU6BBN2/graph.json","fetch_events":"https://pith.science/api/pith-number/PSARXI34XQRDTUVWMKTSU6BBN2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2/action/storage_attestation","attest_author":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2/action/author_attestation","sign_citation":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2/action/citation_signature","submit_replication":"https://pith.science/pith/PSARXI34XQRDTUVWMKTSU6BBN2/action/replication_record"}},"created_at":"2026-07-05T11:34:56.229266+00:00","updated_at":"2026-07-05T11:34:56.229266+00:00"}