{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:64NSQ2DT4UUFWSRHQNGUWUSI3W","short_pith_number":"pith:64NSQ2DT","schema_version":"1.0","canonical_sha256":"f71b286873e5285b4a27834d4b5248dd8ef1f655c68de8d9e60da314ef2e4240","source":{"kind":"arxiv","id":"2505.24445","version":1},"attestation_state":"computed","paper":{"title":"Learning Safety Constraints for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Xin Chen, Yarden As","submitted_at":"2025-05-30T10:30:24Z","abstract_excerpt":"Large language models (LLMs) have emerged as powerful tools but pose significant safety risks through harmful outputs and vulnerability to adversarial attacks. We propose SaP, short for Safety Polytope, a geometric approach to LLM safety that learns and enforces multiple safety constraints directly in the model's representation space. We develop a framework that identifies safe and unsafe regions via the polytope's facets, enabling both detection and correction of unsafe outputs through geometric steering. Unlike existing approaches that modify model weights, SaP operates post-hoc in the repre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24445","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T10:30:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f7abe9442f00a36f3d60b13add733c15db11cada4d085aa91331cabac780f6d1","abstract_canon_sha256":"66bea41146554520d626f4263334fecbe1259ff93893188377b300740ef98c7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:49.925130Z","signature_b64":"cSfUZoF5F/ZgX36hS9Q2JTR/fxfFsd9XlUzzAHQBZhLeHHZFYrhQpBIEISEONcK20WvC1HyBCcYWhBhmSlWpAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f71b286873e5285b4a27834d4b5248dd8ef1f655c68de8d9e60da314ef2e4240","last_reissued_at":"2026-07-05T11:12:49.924574Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:49.924574Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Safety Constraints for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Xin Chen, Yarden As","submitted_at":"2025-05-30T10:30:24Z","abstract_excerpt":"Large language models (LLMs) have emerged as powerful tools but pose significant safety risks through harmful outputs and vulnerability to adversarial attacks. We propose SaP, short for Safety Polytope, a geometric approach to LLM safety that learns and enforces multiple safety constraints directly in the model's representation space. We develop a framework that identifies safe and unsafe regions via the polytope's facets, enabling both detection and correction of unsafe outputs through geometric steering. Unlike existing approaches that modify model weights, SaP operates post-hoc in the repre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24445","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24445/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24445","created_at":"2026-07-05T11:12:49.924638+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24445v1","created_at":"2026-07-05T11:12:49.924638+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24445","created_at":"2026-07-05T11:12:49.924638+00:00"},{"alias_kind":"pith_short_12","alias_value":"64NSQ2DT4UUF","created_at":"2026-07-05T11:12:49.924638+00:00"},{"alias_kind":"pith_short_16","alias_value":"64NSQ2DT4UUFWSRH","created_at":"2026-07-05T11:12:49.924638+00:00"},{"alias_kind":"pith_short_8","alias_value":"64NSQ2DT","created_at":"2026-07-05T11:12:49.924638+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15531","citing_title":"Greedy Coordinate Diffusion: Effective and Semantically Coherent Adversarial Attacks via Diffusion Guidance","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07335","citing_title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2602.03433","citing_title":"When control meets large language models: From words to dynamics","ref_index":257,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16339","citing_title":"Preference Instability in Reward Models: Detection and Mitigation via Sparse Autoencoders","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W","json":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W.json","graph_json":"https://pith.science/api/pith-number/64NSQ2DT4UUFWSRHQNGUWUSI3W/graph.json","events_json":"https://pith.science/api/pith-number/64NSQ2DT4UUFWSRHQNGUWUSI3W/events.json","paper":"https://pith.science/paper/64NSQ2DT"},"agent_actions":{"view_html":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W","download_json":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W.json","view_paper":"https://pith.science/paper/64NSQ2DT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24445&json=true","fetch_graph":"https://pith.science/api/pith-number/64NSQ2DT4UUFWSRHQNGUWUSI3W/graph.json","fetch_events":"https://pith.science/api/pith-number/64NSQ2DT4UUFWSRHQNGUWUSI3W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W/action/storage_attestation","attest_author":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W/action/author_attestation","sign_citation":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W/action/citation_signature","submit_replication":"https://pith.science/pith/64NSQ2DT4UUFWSRHQNGUWUSI3W/action/replication_record"}},"created_at":"2026-07-05T11:12:49.924638+00:00","updated_at":"2026-07-05T11:12:49.924638+00:00"}