{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W5GZYEOFLHFAWAYNNROJHQYRPN","short_pith_number":"pith:W5GZYEOF","schema_version":"1.0","canonical_sha256":"b74d9c11c559ca0b030d6c5c93c3117b44491a6f02bfed97e12923c134e2ece2","source":{"kind":"arxiv","id":"2406.15518","version":1},"attestation_state":"computed","paper":{"title":"Steering Without Side Effects: Improving Post-Deployment Control of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Lyzhov, Asa Cooper Stickland, Jacob Pfau, Salsabila Mahdi, Samuel R. Bowman","submitted_at":"2024-06-21T01:37:39Z","abstract_excerpt":"Language models (LMs) have been shown to behave unexpectedly post-deployment. For example, new jailbreaks continually arise, allowing model misuse, despite extensive red-teaming and adversarial training from developers. Given most model queries are unproblematic and frequent retraining results in unstable user experience, methods for mitigation of worst-case behavior should be targeted. One such method is classifying inputs as potentially problematic, then selectively applying steering vectors on these problematic inputs, i.e. adding particular vectors to model hidden states. However, steering"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15518","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-21T01:37:39Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e92d3cfc52983ed62efd1922d1b004c5c7760b87d6f2fa167ba2e06731bc4716","abstract_canon_sha256":"20f0ac2c19c9b66eddf03162159ed1fd2304084516bd5ac118349695655c1372"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:23.637245Z","signature_b64":"I/tOCeLqRnyoX3zTa2NZw+UV0vbhlR6XeMxgMsXu6axjYorOAH0IqvAgJUGSDvKeniwV6nE7Zkt7sdIMeKZ5Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b74d9c11c559ca0b030d6c5c93c3117b44491a6f02bfed97e12923c134e2ece2","last_reissued_at":"2026-07-05T08:35:23.636756Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:23.636756Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Steering Without Side Effects: Improving Post-Deployment Control of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Lyzhov, Asa Cooper Stickland, Jacob Pfau, Salsabila Mahdi, Samuel R. Bowman","submitted_at":"2024-06-21T01:37:39Z","abstract_excerpt":"Language models (LMs) have been shown to behave unexpectedly post-deployment. For example, new jailbreaks continually arise, allowing model misuse, despite extensive red-teaming and adversarial training from developers. Given most model queries are unproblematic and frequent retraining results in unstable user experience, methods for mitigation of worst-case behavior should be targeted. One such method is classifying inputs as potentially problematic, then selectively applying steering vectors on these problematic inputs, i.e. adding particular vectors to model hidden states. However, steering"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15518","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15518/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15518","created_at":"2026-07-05T08:35:23.636815+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15518v1","created_at":"2026-07-05T08:35:23.636815+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15518","created_at":"2026-07-05T08:35:23.636815+00:00"},{"alias_kind":"pith_short_12","alias_value":"W5GZYEOFLHFA","created_at":"2026-07-05T08:35:23.636815+00:00"},{"alias_kind":"pith_short_16","alias_value":"W5GZYEOFLHFAWAYN","created_at":"2026-07-05T08:35:23.636815+00:00"},{"alias_kind":"pith_short_8","alias_value":"W5GZYEOF","created_at":"2026-07-05T08:35:23.636815+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.00166","citing_title":"Disentangled Safety Adapters Enable Efficient Guardrails and Flexible Inference-Time Alignment","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2508.16846","citing_title":"BASIL: Bayesian Assessment of Sycophancy in LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10467","citing_title":"User Detection and Response Patterns of Sycophantic Behavior in Conversational AI","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05223","citing_title":"Structural Instability of Feature Composition","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN","json":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN.json","graph_json":"https://pith.science/api/pith-number/W5GZYEOFLHFAWAYNNROJHQYRPN/graph.json","events_json":"https://pith.science/api/pith-number/W5GZYEOFLHFAWAYNNROJHQYRPN/events.json","paper":"https://pith.science/paper/W5GZYEOF"},"agent_actions":{"view_html":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN","download_json":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN.json","view_paper":"https://pith.science/paper/W5GZYEOF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15518&json=true","fetch_graph":"https://pith.science/api/pith-number/W5GZYEOFLHFAWAYNNROJHQYRPN/graph.json","fetch_events":"https://pith.science/api/pith-number/W5GZYEOFLHFAWAYNNROJHQYRPN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN/action/storage_attestation","attest_author":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN/action/author_attestation","sign_citation":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN/action/citation_signature","submit_replication":"https://pith.science/pith/W5GZYEOFLHFAWAYNNROJHQYRPN/action/replication_record"}},"created_at":"2026-07-05T08:35:23.636815+00:00","updated_at":"2026-07-05T08:35:23.636815+00:00"}