{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XJLW2WGEXTUBDADXCCO5BTUZGK","short_pith_number":"pith:XJLW2WGE","schema_version":"1.0","canonical_sha256":"ba576d58c4bce8118077109dd0ce9932833ee87ab0f8d1817d8e208e164321b2","source":{"kind":"arxiv","id":"2506.04250","version":1},"attestation_state":"computed","paper":{"title":"SafeSteer: Interpretable Safety Steering with Refusal-Evasion in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amrita Bhattacharjee, Christopher Parisien, Shaona Ghosh, Yftah Ziser","submitted_at":"2025-06-01T01:19:37Z","abstract_excerpt":"Fine-tuning large language models (LLMs) to adapt to evolving safety policies is costly and impractical. Mechanistic interpretability enables inference-time control through latent activation steering, yet its potential for precise, customizable safety adjustments remains largely untapped. This paper investigates an approach called SafeSteer for guiding the outputs of LLMs by: (i) leveraging category-specific steering vectors for more precise control, (ii) employing a simple, gradient-free unsupervised method to enhance safety steering while preserving text quality, topic relevance, and without"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04250","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-01T01:19:37Z","cross_cats_sorted":[],"title_canon_sha256":"0829b3ac5313b4d89d3623c1c92f3f7ae922fa28cb39d6f7a03db8cda6cc2855","abstract_canon_sha256":"58b0cfeb78abe70758779e5a94ded20ee6675e249c32eb5c80690ba9858dbf81"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:11.089007Z","signature_b64":"YW6+duo14IqO3LQKNhov13XRrN/7E7mDFDeOihma7pB3+Y2J4nrVeJiLZKs0wr/l3qdil1Y0B3o4QN6YlFy6DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba576d58c4bce8118077109dd0ce9932833ee87ab0f8d1817d8e208e164321b2","last_reissued_at":"2026-07-05T11:16:11.088499Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:11.088499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SafeSteer: Interpretable Safety Steering with Refusal-Evasion in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amrita Bhattacharjee, Christopher Parisien, Shaona Ghosh, Yftah Ziser","submitted_at":"2025-06-01T01:19:37Z","abstract_excerpt":"Fine-tuning large language models (LLMs) to adapt to evolving safety policies is costly and impractical. Mechanistic interpretability enables inference-time control through latent activation steering, yet its potential for precise, customizable safety adjustments remains largely untapped. This paper investigates an approach called SafeSteer for guiding the outputs of LLMs by: (i) leveraging category-specific steering vectors for more precise control, (ii) employing a simple, gradient-free unsupervised method to enhance safety steering while preserving text quality, topic relevance, and without"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04250","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04250/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04250","created_at":"2026-07-05T11:16:11.088563+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04250v1","created_at":"2026-07-05T11:16:11.088563+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04250","created_at":"2026-07-05T11:16:11.088563+00:00"},{"alias_kind":"pith_short_12","alias_value":"XJLW2WGEXTUB","created_at":"2026-07-05T11:16:11.088563+00:00"},{"alias_kind":"pith_short_16","alias_value":"XJLW2WGEXTUBDADX","created_at":"2026-07-05T11:16:11.088563+00:00"},{"alias_kind":"pith_short_8","alias_value":"XJLW2WGE","created_at":"2026-07-05T11:16:11.088563+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.20766","citing_title":"Turning the Spell Around: Lightweight Alignment Amplification via Rank-One Safety Injection","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK","json":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK.json","graph_json":"https://pith.science/api/pith-number/XJLW2WGEXTUBDADXCCO5BTUZGK/graph.json","events_json":"https://pith.science/api/pith-number/XJLW2WGEXTUBDADXCCO5BTUZGK/events.json","paper":"https://pith.science/paper/XJLW2WGE"},"agent_actions":{"view_html":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK","download_json":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK.json","view_paper":"https://pith.science/paper/XJLW2WGE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04250&json=true","fetch_graph":"https://pith.science/api/pith-number/XJLW2WGEXTUBDADXCCO5BTUZGK/graph.json","fetch_events":"https://pith.science/api/pith-number/XJLW2WGEXTUBDADXCCO5BTUZGK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK/action/storage_attestation","attest_author":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK/action/author_attestation","sign_citation":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK/action/citation_signature","submit_replication":"https://pith.science/pith/XJLW2WGEXTUBDADXCCO5BTUZGK/action/replication_record"}},"created_at":"2026-07-05T11:16:11.088563+00:00","updated_at":"2026-07-05T11:16:11.088563+00:00"}