{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:RKAT2B7N3USRWPFIO5NUMPM2US","short_pith_number":"pith:RKAT2B7N","schema_version":"1.0","canonical_sha256":"8a813d07eddd251b3ca8775b463d9aa4bcd9fa52de5b743e57a709fc685d66a6","source":{"kind":"arxiv","id":"2607.20286","version":1},"attestation_state":"computed","paper":{"title":"Sound Probabilistic Safety Bounds for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alessandro Abate, Anne-Kathrin Schmuck, Mahdi Nazeri, Sadegh Soudjani","submitted_at":"2026-07-22T15:31:28Z","abstract_excerpt":"We propose a novel framework for computing rigorous bounds on the probability that a large language model (LLM) generates harmful output to a given prompt. We study a new application of the Clopper-Pearson confidence intervals to obtain probably approximately correct (PAC) bounds for this problem. As our main technical contribution, we propose an algorithm that leverages features in the latent space to prioritize exploring branches in the auto-regressive generation tree that are more likely to produce harmful outputs. Our approach in particular enables the efficient computation of useful lower"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.20286","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-22T15:31:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"439785a5f6cf456080e015c651ebe2da352ee78b5f39a366e4b9c5b63af206f3","abstract_canon_sha256":"eae982be25d36d3857d48e173768db0c624c8a5b3ed384864b5f7a8d8fe8beb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-23T01:25:12.393969Z","signature_b64":"VI/Fo179o8UzYX5bljRCG0F0noJNBTgDcA4Z3fSETzbQ/9H70VVyQUn++EXFzrrKHitrBY02x9xhUNRF1u7KBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a813d07eddd251b3ca8775b463d9aa4bcd9fa52de5b743e57a709fc685d66a6","last_reissued_at":"2026-07-23T01:25:12.393154Z","signature_status":"signed_v1","first_computed_at":"2026-07-23T01:25:12.393154Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sound Probabilistic Safety Bounds for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alessandro Abate, Anne-Kathrin Schmuck, Mahdi Nazeri, Sadegh Soudjani","submitted_at":"2026-07-22T15:31:28Z","abstract_excerpt":"We propose a novel framework for computing rigorous bounds on the probability that a large language model (LLM) generates harmful output to a given prompt. We study a new application of the Clopper-Pearson confidence intervals to obtain probably approximately correct (PAC) bounds for this problem. As our main technical contribution, we propose an algorithm that leverages features in the latent space to prioritize exploring branches in the auto-regressive generation tree that are more likely to produce harmful outputs. Our approach in particular enables the efficient computation of useful lower"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.20286","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.20286/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.20286","created_at":"2026-07-23T01:25:12.393575+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.20286v1","created_at":"2026-07-23T01:25:12.393575+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.20286","created_at":"2026-07-23T01:25:12.393575+00:00"},{"alias_kind":"pith_short_12","alias_value":"RKAT2B7N3USR","created_at":"2026-07-23T01:25:12.393575+00:00"},{"alias_kind":"pith_short_16","alias_value":"RKAT2B7N3USRWPFI","created_at":"2026-07-23T01:25:12.393575+00:00"},{"alias_kind":"pith_short_8","alias_value":"RKAT2B7N","created_at":"2026-07-23T01:25:12.393575+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US","json":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US.json","graph_json":"https://pith.science/api/pith-number/RKAT2B7N3USRWPFIO5NUMPM2US/graph.json","events_json":"https://pith.science/api/pith-number/RKAT2B7N3USRWPFIO5NUMPM2US/events.json","paper":"https://pith.science/paper/RKAT2B7N"},"agent_actions":{"view_html":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US","download_json":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US.json","view_paper":"https://pith.science/paper/RKAT2B7N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.20286&json=true","fetch_graph":"https://pith.science/api/pith-number/RKAT2B7N3USRWPFIO5NUMPM2US/graph.json","fetch_events":"https://pith.science/api/pith-number/RKAT2B7N3USRWPFIO5NUMPM2US/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US/action/storage_attestation","attest_author":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US/action/author_attestation","sign_citation":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US/action/citation_signature","submit_replication":"https://pith.science/pith/RKAT2B7N3USRWPFIO5NUMPM2US/action/replication_record"}},"created_at":"2026-07-23T01:25:12.393575+00:00","updated_at":"2026-07-23T01:25:12.393575+00:00"}